{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":552,"total_is_capped":false,"direct_labels_cover":2,"predictions_cover":552,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"b13bc4f485d2","filters":{"topic":"Advanced Text Analysis Techniques"}},"results":[{"id":"W2561541699","doi":"10.1111/isj.12131","title":"Minimum sample size estimation in PLS‐SEM: The inverse square root and gamma‐exponential methods","year":2016,"lang":"en","type":"article","venue":"Information Systems Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2185,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Square root; Mathematics; Inverse; Statistics; Sample size determination; Multivariate statistics; Monte Carlo method; Exponential function; Applied mathematics; Mean squared error; Mathematical optimization; Mathematical analysis","authors":[{"name":"Ned Kock","is_ca":false},{"name":"Pierre Hadaya","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01700328630793958,"gpt":0.3136178639966472,"spread":0.2966145776887076,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04549818,0.001429323,0.001455556,0.001840895,0.0008694705,0.002137095,0.002067225,0.001946426,0.004242044],"category_scores_gemma":[0.170002,0.000744589,0.001478471,0.002169561,0.002541301,0.003142085,0.002590413,0.003057126,0.001065389],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001077713,"about_ca_system_score_gemma":0.00197953,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001751755,"about_ca_topic_score_gemma":0.00224885,"domain_scores_codex":[0.9617673,0.03242719,0.0007233765,0.001646958,0.003214587,0.0002206711],"domain_scores_gemma":[0.8605692,0.1209759,0.004057577,0.007596064,0.006394066,0.0004070862],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009782168,0.0003900697,0.01059586,0.001358024,0.0006084145,0.0003341831,0.002265401,0.1835053,0.006186608,0.1832291,0.007773785,0.602775],"study_design_scores_gemma":[0.0002075875,0.0004518339,0.00522238,0.0003294137,0.0001638214,0.0003035231,0.0004177908,0.761536,0.006848429,0.2165294,0.007823014,0.0001667833],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009443181,0.0002350109,0.9887319,0.0002518641,0.0000383202,0.0001996095,0.00006390877,0.0002081234,0.0008281395],"genre_scores_gemma":[0.1631953,0.0003244158,0.8342713,0.0001498971,0.00003707267,0.0008451158,0.0001684299,0.0002372292,0.0007711863],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04549818,"threshold_uncertainty_score":0.2406203,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2895553377","doi":"10.21105/joss.00774","title":"quanteda: An R package for the quantitative analysis of textual data","year":2018,"lang":"en","type":"article","venue":"The Journal of Open Source Software","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1323,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"De Beers (Canada)","funders":"London School of Economics and Political Science","keywords":"R package; Computer science; Natural language processing; Information retrieval; Programming language","authors":[{"name":"Kenneth Benoit","is_ca":false},{"name":"Kohei Watanabe","is_ca":false},{"name":"H. P. Wang","is_ca":true},{"name":"Paul Nulty","is_ca":false},{"name":"Adam Obeng","is_ca":false},{"name":"Stefan Müller","is_ca":false},{"name":"Akitaka Matsuo","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1220943930715411,"gpt":0.4211929597768534,"spread":0.2990985667053123,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006707033,0.002441942,0.002382296,0.004722014,0.0008669276,0.004059064,0.00294895,0.0008442505,0.07443252],"category_scores_gemma":[0.06362126,0.001460198,0.002594867,0.004280706,0.001017138,0.002820018,0.003058546,0.002848078,0.04489454],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007382481,"about_ca_system_score_gemma":0.003106107,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002930819,"about_ca_topic_score_gemma":0.004392474,"domain_scores_codex":[0.9946527,0.002297923,0.0005183062,0.001132652,0.001216435,0.0001819919],"domain_scores_gemma":[0.9680011,0.02364318,0.002284335,0.003285485,0.002339555,0.000446265],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007693812,0.0001001968,0.01101151,0.007867895,0.003391676,0.0005506496,0.0008980616,0.008467674,0.006437853,0.01884106,0.8058477,0.1358163],"study_design_scores_gemma":[0.0004562131,0.0001991877,0.01804657,0.000929651,0.001225604,0.001050459,0.0003217137,0.04899456,0.009679314,0.08912782,0.8295979,0.0003709803],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.005810041,0.001439445,0.4754369,0.001159322,0.0006601073,0.0007521333,0.3189017,0.188569,0.007271363],"genre_scores_gemma":[0.04798994,0.001318781,0.614308,0.00140266,0.0004191116,0.006595136,0.1711677,0.1455392,0.01125947],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.07443252,"threshold_uncertainty_score":0.2490016,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1494478641","doi":"10.1007/978-0-387-69810-6","title":"The Statistical Analysis of Recurrent Events","year":2007,"lang":"en","type":"book","venue":"Statistics for biology and health","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":738,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science","authors":[{"name":"Richard J. Cook","is_ca":true},{"name":"Jerald F. Lawless","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04241724093842392,"gpt":0.4368227611509031,"spread":0.3944055202124792,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006793679,0.001476457,0.002170763,0.003147613,0.000371675,0.002201007,0.002010452,0.001172129,0.006422818],"category_scores_gemma":[0.03536831,0.0009659721,0.001598893,0.003766927,0.002327925,0.0029737,0.001179722,0.003842383,0.005357296],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007030325,"about_ca_system_score_gemma":0.001114642,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009019421,"about_ca_topic_score_gemma":0.001179103,"domain_scores_codex":[0.9956585,0.002000228,0.0003433472,0.0006231916,0.001286121,0.00008851253],"domain_scores_gemma":[0.9694932,0.0257557,0.0007567508,0.002493105,0.001304283,0.0001969307],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00007544936,0.00004787039,0.001284272,0.0008406391,0.0003333868,0.0002225916,0.0003034676,0.01475972,0.002443078,0.2289163,0.09968697,0.6510864],"study_design_scores_gemma":[0.00002509755,0.00005879429,0.002439295,0.0001675868,0.0001144288,0.0006036725,0.00005395489,0.1153194,0.001896795,0.8247179,0.05452982,0.0000733073],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0005392426,0.005261445,0.9900244,0.0004460651,0.0003603286,0.00002373664,0.0003590423,0.001142812,0.001842854],"genre_scores_gemma":[0.04111317,0.01216979,0.9222741,0.0009086003,0.00288598,0.0006486338,0.00345956,0.001937338,0.01460275],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006793679,"threshold_uncertainty_score":0.03592885,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2072435181","doi":"10.1207/s1532690xci2402_3","title":"Helping Students Understand Challenging Topics in Science Through Ontology Training","year":2006,"lang":"en","type":"article","venue":"Cognition and Instruction","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":285,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"Spencer Foundation; Andrew W. Mellon Foundation","keywords":"Ontology; Conceptual change; Science education; Task (project management); Mathematics education; Computer science; Concept learning; Psychology; Epistemology; Engineering","authors":[{"name":"James D. Slotta","is_ca":true},{"name":"T. H. Michelene","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04250187749906544,"gpt":0.3179108858070179,"spread":0.2754090083079525,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001919808,0.0007374318,0.0004284689,0.001030024,0.0006951456,0.001884996,0.001028686,0.001001519,0.005177291],"category_scores_gemma":[0.01219942,0.0002628619,0.000455368,0.0005052235,0.0009052248,0.003566552,0.002468814,0.001547104,0.001388269],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005128833,"about_ca_system_score_gemma":0.001279149,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005446271,"about_ca_topic_score_gemma":0.001129714,"domain_scores_codex":[0.9992399,0.000237749,0.00005120466,0.0001662241,0.00018561,0.0001192569],"domain_scores_gemma":[0.9945585,0.003298255,0.0006429639,0.0005141111,0.0004309502,0.0005552832],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0004180742,0.01363169,0.046858,0.001048289,0.00004974756,0.001058004,0.06783462,0.003514779,0.09138113,0.01700026,0.01744765,0.739758],"study_design_scores_gemma":[0.001012438,0.007168055,0.1292216,0.001080532,0.0004502536,0.004632069,0.07682286,0.06989305,0.1659147,0.2241399,0.3192632,0.0004013357],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9336768,0.0002046796,0.05126131,0.002379022,0.00007126378,0.0004517565,0.00007487534,0.0007500224,0.01113025],"genre_scores_gemma":[0.8601751,0.0007236146,0.1294516,0.000835301,0.0000442996,0.0005298897,0.0003265517,0.00008277627,0.007830862],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005177291,"threshold_uncertainty_score":0.01731974,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1544240449","doi":"10.1007/3-540-45486-1_4","title":"Using Noun Phrase Heads to Extract Document Keyphrases","year":2000,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":236,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Noun phrase; Automatic summarization; Natural language processing; Artificial intelligence; Phrase; Task (project management); Extractor; Head (geology); Noun; Proper noun; Information retrieval; Linguistics","authors":[{"name":"Ken Barker","is_ca":true},{"name":"Nadia Cornacchia","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02587671481000824,"gpt":0.3118338681191701,"spread":0.2859571533091619,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007002905,0.00195195,0.001196959,0.006049979,0.001044377,0.002798881,0.0007317809,0.001057562,0.01189557],"category_scores_gemma":[0.003669307,0.0008113304,0.001060241,0.004638606,0.000534061,0.003023813,0.001279289,0.001326145,0.0214102],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006419757,"about_ca_system_score_gemma":0.001543352,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002821813,"about_ca_topic_score_gemma":0.003976644,"domain_scores_codex":[0.9993348,0.00005795266,0.00008623162,0.0002013485,0.0002372135,0.00008243225],"domain_scores_gemma":[0.9964753,0.001404814,0.0003102396,0.0003016573,0.001356605,0.0001513584],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006315422,0.0001122363,0.002544596,0.001735001,0.00009180542,0.001176117,0.0008391306,0.0008613307,0.2014068,0.004592314,0.02451679,0.7614924],"study_design_scores_gemma":[0.000301813,0.0009526252,0.02604275,0.000647313,0.0008528254,0.005216947,0.003462117,0.1082341,0.5635957,0.02995366,0.2602973,0.0004428894],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06625529,0.003490966,0.8425766,0.0005862986,0.0008535942,0.0009565554,0.01765587,0.05235197,0.01527291],"genre_scores_gemma":[0.1452871,0.002840224,0.8003367,0.0002240728,0.0004122438,0.0004819517,0.02872278,0.00375346,0.01794153],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01189557,"threshold_uncertainty_score":0.03979462,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2148404145","doi":"10.3115/v1/d14-1168","title":"Abstractive Summarization of Product Reviews Using Discourse Structure","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":200,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Automatic summarization; Computer science; Natural language processing; Product (mathematics); Linguistics; Artificial intelligence; Mathematics; Philosophy","authors":[{"name":"Shima Gerani","is_ca":true},{"name":"Yashar Mehdad","is_ca":true},{"name":"Giuseppe Carenini","is_ca":true},{"name":"Raymond T. Ng","is_ca":true},{"name":"Bita Nejat","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01939090081418264,"gpt":0.326621961672267,"spread":0.3072310608580843,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001897499,0.001884028,0.00130155,0.006441223,0.0007875125,0.00295382,0.001323418,0.0009060662,0.002957125],"category_scores_gemma":[0.01080162,0.0006405336,0.001068269,0.003065357,0.0003934974,0.003851225,0.001404253,0.001186086,0.002579065],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007580305,"about_ca_system_score_gemma":0.001293221,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002387718,"about_ca_topic_score_gemma":0.003325521,"domain_scores_codex":[0.9977144,0.0007214856,0.0002533766,0.0005553621,0.0006858599,0.00006954461],"domain_scores_gemma":[0.9927004,0.003334705,0.00112337,0.0005770989,0.002137166,0.0001272358],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003025225,0.0001741771,0.001546536,0.001827616,0.0002168757,0.0004304475,0.002566661,0.01340736,0.05523998,0.006826086,0.02538401,0.8920778],"study_design_scores_gemma":[0.0002170655,0.0006730478,0.007032035,0.0005131169,0.0009761709,0.000787688,0.002576258,0.6140401,0.1426519,0.04254547,0.1877068,0.0002803969],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.033097,0.001708925,0.938297,0.0009850557,0.0002831234,0.0006986326,0.003331871,0.01762539,0.003973037],"genre_scores_gemma":[0.1132914,0.001142723,0.8689172,0.0002075729,0.000399634,0.000467379,0.01057252,0.0007954981,0.004205952],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006441223,"threshold_uncertainty_score":0.01003504,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2946989286","doi":"10.1007/s41237-019-00085-5","title":"A concept analysis of methodological research on composite-based structural equation modeling: bridging PLSPM and GSCA","year":2019,"lang":"en","type":"article","venue":"Behaviormetrika","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":191,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"","keywords":"Structural equation modeling; Path analysis (statistics); Computer science; Bridging (networking); Econometrics; Mathematics; Machine learning","authors":[{"name":"Heungsun Hwang","is_ca":true},{"name":"Marko Sarstedt","is_ca":false},{"name":"Jun‐Hwa Cheah","is_ca":false},{"name":"Christian M. Ringle","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3321505942566969,"gpt":0.4802027912565942,"spread":0.1480521969998973,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04648649,0.001255621,0.001298563,0.01161316,0.00321113,0.006694808,0.002467193,0.001588809,0.00697458],"category_scores_gemma":[0.106726,0.0007771772,0.002359787,0.01116284,0.01079332,0.0108018,0.004519018,0.004023172,0.0005498477],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005687228,"about_ca_system_score_gemma":0.01077011,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002793974,"about_ca_topic_score_gemma":0.00276325,"domain_scores_codex":[0.9618052,0.02891259,0.001480703,0.003116968,0.004218023,0.0004666259],"domain_scores_gemma":[0.855663,0.1217324,0.004163746,0.007173066,0.01045035,0.0008174258],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00003518818,0.00007189994,0.00237046,0.0004304424,0.00008256981,0.000037383,0.002713644,0.001107296,0.0003767933,0.9534822,0.0009508886,0.03834114],"study_design_scores_gemma":[0.00006819316,0.0001911986,0.005131299,0.0009528486,0.0002097719,0.0001834191,0.005243569,0.03688112,0.001507144,0.9330151,0.01655585,0.00006055175],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02680905,0.001409782,0.9537502,0.004612106,0.0002299973,0.0006126327,0.0003028087,0.0001858688,0.01208755],"genre_scores_gemma":[0.3208615,0.0009423307,0.6733228,0.0008927,0.0001593157,0.002208093,0.0002736521,0.0001148282,0.001224856],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9535135,"threshold_uncertainty_score":0.2458469,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2230887875","doi":"10.1080/10888438.2015.1107073","title":"The Random Forests statistical technique: An examination of its value for the study of reading","year":2016,"lang":"en","type":"article","venue":"Scientific Studies of Reading","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":168,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"McMaster University","funders":"Eunice Kennedy Shriver National Institute of Child Health and Human Development; National Institute of Child Health and Human Development; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Reading (process); Computer science; Random forest; Statistical analysis; Value (mathematics); Statistics; Artificial intelligence; Machine learning; Mathematics; Linguistics","authors":[{"name":"Kazunaga Matsuki","is_ca":true},{"name":"Victor Kuperman","is_ca":true},{"name":"Julie A. Van Dyke","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04712200114569937,"gpt":0.3674317793021761,"spread":0.3203097781564768,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1156854,0.001468492,0.002162835,0.005674127,0.001503112,0.002490606,0.001703435,0.00182829,0.00243917],"category_scores_gemma":[0.2444715,0.0005647267,0.001884998,0.006434824,0.002702569,0.003723353,0.001796629,0.003559293,0.0006430743],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000891483,"about_ca_system_score_gemma":0.002811923,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00564093,"about_ca_topic_score_gemma":0.007270517,"domain_scores_codex":[0.9546769,0.03651902,0.001113534,0.00206305,0.005269033,0.0003585364],"domain_scores_gemma":[0.5790308,0.3993357,0.005000676,0.008267893,0.007350882,0.001014205],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0008311365,0.0002468865,0.08258685,0.001687871,0.002353222,0.0006815543,0.003680529,0.06175264,0.00262362,0.09100565,0.01331787,0.7392322],"study_design_scores_gemma":[0.0002710486,0.001260592,0.06560376,0.002714894,0.0009141791,0.002457146,0.002110597,0.5126429,0.003654337,0.366918,0.04095934,0.0004932142],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03267493,0.006465254,0.9527307,0.002163862,0.0003381699,0.0003618429,0.0004608508,0.0008896645,0.003914651],"genre_scores_gemma":[0.299451,0.003518767,0.6933197,0.00061108,0.0004284533,0.0007129695,0.0003683944,0.000603381,0.000986213],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8843147,"threshold_uncertainty_score":0.61181,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1970043789","doi":"10.1145/1645953.1646076","title":"Detecting topic evolution in scientific literature","year":2009,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":165,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"National Science Foundation","keywords":"Latent Dirichlet allocation; Computer science; Topic model; Data science; Citation; Inheritance (genetic algorithm); Scientific literature; Information retrieval; World Wide Web","authors":[{"name":"Qi He","is_ca":false},{"name":"Bi Yu Chen","is_ca":false},{"name":"Jian Pei","is_ca":true},{"name":"Baojun Qiu","is_ca":false},{"name":"Prasenjit Mitra","is_ca":false},{"name":"C. Lee Giles","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.006675345525235707,"gpt":0.262146813315761,"spread":0.2554714677905253,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.007704913,0.0005139778,0.001038888,0.01741944,0.001269754,0.003117161,0.00122204,0.001999363,0.0006122695],"category_scores_gemma":[0.04195796,0.0004555421,0.001003273,0.01380992,0.0007417907,0.005199559,0.002047759,0.001253953,0.0004997904],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001394564,"about_ca_system_score_gemma":0.001118589,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003022837,"about_ca_topic_score_gemma":0.002856842,"domain_scores_codex":[0.9962012,0.001058444,0.0003842993,0.0009771591,0.001129224,0.0002495946],"domain_scores_gemma":[0.9653881,0.02216405,0.005761709,0.001517864,0.004399699,0.0007685349],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004658366,0.0003323677,0.3185823,0.001080572,0.0005162393,0.0008286027,0.005629165,0.03614933,0.02016827,0.01645765,0.008023629,0.5917659],"study_design_scores_gemma":[0.00009140813,0.0003197289,0.2705837,0.0002692192,0.0006392265,0.002359456,0.002446618,0.599698,0.02118755,0.07988407,0.02231139,0.0002096504],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8046042,0.009264443,0.1783381,0.001114076,0.000184966,0.0002021928,0.001248211,0.001198614,0.003845332],"genre_scores_gemma":[0.9367045,0.001884067,0.05739059,0.0001126981,0.0003683979,0.0001528099,0.001968283,0.00009092713,0.001327833],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9825805,"threshold_uncertainty_score":0.04074794,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1486865875","doi":"10.48550/arxiv.cs/0308033","title":"Coherent Keyphrase Extraction via Web Mining","year":2003,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":160,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Extraction (chemistry); Web mining; World Wide Web; Web page","authors":[{"name":"Peter D. Turney","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04399676745781409,"gpt":0.3191097352841954,"spread":0.2751129678263813,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002348078,0.001611956,0.001339529,0.01308089,0.001206836,0.002983325,0.001472895,0.001349852,0.003104736],"category_scores_gemma":[0.01564712,0.0007922284,0.001548983,0.01069909,0.0008047823,0.004820709,0.002142414,0.001305333,0.004792546],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000682753,"about_ca_system_score_gemma":0.001535089,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001504788,"about_ca_topic_score_gemma":0.001909236,"domain_scores_codex":[0.9970228,0.0005550193,0.0004171316,0.0008034849,0.001010289,0.0001912325],"domain_scores_gemma":[0.9910777,0.00375469,0.00130085,0.00154199,0.002156946,0.0001677708],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000436544,0.00022886,0.004738349,0.001168351,0.0001739014,0.0008314447,0.0005954363,0.007369768,0.06603757,0.009141454,0.01055716,0.8987212],"study_design_scores_gemma":[0.0002795307,0.0004848427,0.01587915,0.0004215928,0.0005277327,0.003632981,0.001743796,0.4729874,0.2514022,0.1298884,0.1224449,0.0003074726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03139867,0.001320103,0.9532977,0.000336591,0.00008056884,0.0005689046,0.002099376,0.008277277,0.002620727],"genre_scores_gemma":[0.123246,0.001027868,0.8659865,0.0001200852,0.0001225743,0.0003417538,0.006132848,0.0005982541,0.002424122],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01308089,"threshold_uncertainty_score":0.01241797,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2163284121","doi":"10.1080/07421222.2000.11045646","title":"The Use of Explanations in Knowledge-Based Systems: Cognitive Perspectives and a Process-Tracing Analysis","year":2000,"lang":"en","type":"article","venue":"Journal of Management Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":144,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Protocol analysis; Cognition; Comprehension; Process tracing; Process (computing); Knowledge management; Cognitive psychology; Computer science; Exploratory analysis; Tracing; Psychology; Qualitative analysis; Data science; Qualitative research; Cognitive science","authors":[{"name":"Izak Benbasat Ji-Ye Mao","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02140219480498735,"gpt":0.289623304492584,"spread":0.2682211096875967,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02012989,0.0008722932,0.000442912,0.006188627,0.001545856,0.005860789,0.001476424,0.002003348,0.001119804],"category_scores_gemma":[0.07860107,0.0006046296,0.001067027,0.002807876,0.00554059,0.008002113,0.002582799,0.001522551,0.0001307954],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002584267,"about_ca_system_score_gemma":0.00191337,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002503251,"about_ca_topic_score_gemma":0.002049671,"domain_scores_codex":[0.9753984,0.01776651,0.001068014,0.0009534999,0.003993463,0.0008200621],"domain_scores_gemma":[0.7954247,0.1862165,0.007699149,0.003991122,0.005969279,0.0006991653],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0006374874,0.0003965838,0.08088252,0.001357766,0.0001936828,0.001557228,0.5974468,0.007062715,0.01507108,0.111013,0.0004350685,0.183946],"study_design_scores_gemma":[0.0003346054,0.001571947,0.1384129,0.00368647,0.0009498251,0.003179275,0.3703684,0.1299408,0.06427374,0.2483485,0.03837846,0.0005550721],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7542848,0.0009668987,0.2336586,0.001371387,0.00001794435,0.0003544577,0.00008353901,0.0001893142,0.009073031],"genre_scores_gemma":[0.9627227,0.000247557,0.0363978,0.00003606214,0.00000862358,0.0001129395,0.00005015579,0.00002290352,0.0004013441],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02012989,"threshold_uncertainty_score":0.1064583,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3125334212","doi":"10.17705/1jais.00237","title":"A Theory-Driven Design Framework for Social Recommender Systems","year":2010,"lang":"en","type":"article","venue":"Journal of the Association for Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":138,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; City University of New York","keywords":"Recommender system; Computer science; Competence (human resources); Similarity (geometry); Designtheory; Design science; Artificial intelligence; Information retrieval; Knowledge management; Human–computer interaction; Psychology; Social psychology","authors":[{"name":"Ofer Arazy","is_ca":true},{"name":"Nanda Kumar","is_ca":false},{"name":"Bracha Shapira","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02622215138912869,"gpt":0.3018622921912353,"spread":0.2756401408021066,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01897665,0.002319627,0.001288241,0.00288882,0.001730853,0.004889253,0.004368645,0.003842135,0.005268858],"category_scores_gemma":[0.01969655,0.001520403,0.002636997,0.001660215,0.004531102,0.003273035,0.003538359,0.003570933,0.001685129],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003393784,"about_ca_system_score_gemma":0.006045831,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003515131,"about_ca_topic_score_gemma":0.004986202,"domain_scores_codex":[0.9846144,0.009619195,0.001032521,0.001291798,0.003037313,0.0004047191],"domain_scores_gemma":[0.9855565,0.008594142,0.0008917573,0.001583497,0.002790847,0.0005833025],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001034649,0.0003428441,0.001274344,0.001250294,0.0002402584,0.0004943987,0.002440951,0.1275821,0.004246604,0.7918071,0.002772258,0.06744541],"study_design_scores_gemma":[0.0003140557,0.0005127887,0.0003434096,0.0004894065,0.0001842274,0.0003470029,0.0006285443,0.4615916,0.002810058,0.4819431,0.05072738,0.000108316],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001147012,0.0001292772,0.9947159,0.0007013702,0.00004173315,0.0005951833,0.00004274834,0.0001172267,0.002509509],"genre_scores_gemma":[0.04326951,0.0002338052,0.9519433,0.0002316775,0.00004225729,0.002925584,0.0001135547,0.00003659966,0.00120369],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01897665,"threshold_uncertainty_score":0.1003593,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2336430626","doi":"10.1037/apl0000108","title":"Initial investigation into computer scoring of candidate essays for personnel selection.","year":2016,"lang":"en","type":"article","venue":"Journal of Applied Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":134,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Campion College","funders":"","keywords":"PsycINFO; Dilemma; Notice; Leverage (statistics); Computer science; Data science; Psychology; Predictive validity; Context (archaeology); Scale (ratio); Selection (genetic algorithm); Personnel selection; Disadvantage; Applied psychology; Social psychology; Artificial intelligence; MEDLINE; Management; Clinical psychology","authors":[{"name":"Michael C. Campion","is_ca":false},{"name":"Michael A. Campion","is_ca":false},{"name":"Emily D. Campion","is_ca":false},{"name":"Matthew H. Reider","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02172397624619364,"gpt":0.3300711010410206,"spread":0.308347124794827,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02731844,0.0006412263,0.0003447861,0.002585573,0.001334231,0.002433471,0.001265528,0.0008376648,0.00448447],"category_scores_gemma":[0.2003808,0.0002764456,0.00031019,0.002592504,0.0007784082,0.001620905,0.001017183,0.001132684,0.001886853],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001504392,"about_ca_system_score_gemma":0.002050103,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003559061,"about_ca_topic_score_gemma":0.007487069,"domain_scores_codex":[0.9793242,0.01396309,0.00121351,0.0008930965,0.004229224,0.000376839],"domain_scores_gemma":[0.760054,0.1695914,0.00934851,0.009653633,0.04895727,0.002395207],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002377022,0.002926945,0.2409893,0.0005850912,0.00009933409,0.0006964975,0.01426698,0.002484714,0.01014019,0.004397934,0.03017956,0.6908564],"study_design_scores_gemma":[0.0006347623,0.009481611,0.7183666,0.0008540137,0.0002116103,0.002131298,0.02649441,0.09010796,0.02653374,0.007116051,0.1178037,0.0002643193],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.928921,0.0006544646,0.03181351,0.003551288,0.0005030642,0.002299297,0.001074237,0.001035504,0.03014767],"genre_scores_gemma":[0.9367945,0.0003705369,0.05140936,0.0006384025,0.0001868762,0.0008716279,0.0009403581,0.0001206038,0.008667771],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02731844,"threshold_uncertainty_score":0.1444754,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2345836734","doi":"10.3758/s13423-016-1053-2","title":"The principals of meaning: Extracting semantic dimensions from co-occurrence models of semantics","year":2016,"lang":"en","type":"review","venue":"Psychonomic Bulletin & Review","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":129,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Semantics (computer science); Meaning (existential); Linguistics; Cognitive psychology; Natural language processing; Cognitive science; Epistemology; Programming language; Computer science; Psychotherapist; Philosophy","authors":[{"name":"Geoff Hollis","is_ca":true},{"name":"Chris Westbury","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07264841213761936,"gpt":0.3794374996201949,"spread":0.3067890874825755,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001933259,0.001202298,0.001203308,0.008381238,0.0004588169,0.002603179,0.001358602,0.0007728176,0.001713027],"category_scores_gemma":[0.008362569,0.0004674766,0.001571704,0.009149184,0.001306502,0.006156547,0.001258442,0.00155703,0.001391599],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007104997,"about_ca_system_score_gemma":0.002390201,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002417253,"about_ca_topic_score_gemma":0.003551598,"domain_scores_codex":[0.9991639,0.0002037038,0.00009925573,0.0001516516,0.0003439711,0.00003763158],"domain_scores_gemma":[0.9962845,0.002439925,0.0003306193,0.0002150991,0.0006590952,0.00007072277],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001949255,0.00009021748,0.009080021,0.004894876,0.0004503408,0.0001239974,0.0007769757,0.00145971,0.002326839,0.02608448,0.01307186,0.9414458],"study_design_scores_gemma":[0.0001686828,0.0002728623,0.06064935,0.005981073,0.002804284,0.002022212,0.004177792,0.06221381,0.00905074,0.5626732,0.2896394,0.000346615],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.06867425,0.4462195,0.4416104,0.009688376,0.001187468,0.0005421393,0.01148817,0.002467043,0.01812272],"genre_scores_gemma":[0.444523,0.2771603,0.2578837,0.0007106393,0.001283012,0.0006444208,0.01414854,0.0003889768,0.003257412],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.008381238,"threshold_uncertainty_score":0.01022416,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2040027527","doi":"10.3758/s13423-015-0808-5","title":"A rational model of function learning","year":2015,"lang":"en","type":"review","venue":"Psychonomic Bulletin & Review","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":112,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Air Force Office of Scientific Research; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Similarity (geometry); Associative learning; Associative property; Psychology; Artificial intelligence; Function (biology); Equivalence (formal languages); Probabilistic logic; Regression; Machine learning; Rational function; Regression analysis; Computer science; Cognitive psychology; Mathematics","authors":[{"name":"Christopher G. Lucas","is_ca":false},{"name":"Thomas L. Griffiths","is_ca":false},{"name":"Joseph Jay Williams","is_ca":false},{"name":"Michael L. Kalish","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08088215470050462,"gpt":0.3658428833187138,"spread":0.2849607286182092,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001356154,0.0007287852,0.0007792951,0.0009701753,0.0003627745,0.001807314,0.001316788,0.001331671,0.006045928],"category_scores_gemma":[0.003423906,0.0002102276,0.0006933503,0.0008755018,0.002994275,0.003156775,0.0007470353,0.001491083,0.002158869],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001683525,"about_ca_system_score_gemma":0.001910717,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002893114,"about_ca_topic_score_gemma":0.001228775,"domain_scores_codex":[0.9993883,0.0002164038,0.00003192403,0.0001205805,0.0001745001,0.00006824236],"domain_scores_gemma":[0.9990375,0.0006032981,0.00006905994,0.0001003206,0.0001592182,0.00003060807],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00003075397,0.00002288892,0.0002920461,0.0002912593,0.00003022009,0.00004734898,0.00005664894,0.006718756,0.0002737608,0.8678882,0.005304125,0.1190439],"study_design_scores_gemma":[0.00001703362,0.00002553744,0.0002437491,0.0001149696,0.00001967219,0.00007405978,0.00002659798,0.0161451,0.0002737497,0.9591392,0.02390882,0.00001149333],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.01985894,0.0921886,0.7355711,0.02568218,0.0008144929,0.0001313962,0.0004619994,0.0008022142,0.124489],"genre_scores_gemma":[0.7279528,0.1007966,0.1090646,0.00452362,0.001425623,0.0004168244,0.0006271027,0.00017049,0.05502241],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.006045928,"threshold_uncertainty_score":0.0202257,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2111310810","doi":"10.1145/1148170.1148262","title":"Statistical precision of information retrieval evaluation","year":2006,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Data mining; Concordance; Population; Information retrieval; Confidence interval; Statistics; Degree (music); Statistical hypothesis testing; Data collection; Test (biology); Artificial intelligence; Mathematics","authors":[{"name":"Gordon V. Cormack","is_ca":true},{"name":"Thomas R. Lynam","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01181805756965918,"gpt":0.3080099634055263,"spread":0.2961919058358671,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1368204,0.001732465,0.003367085,0.01227101,0.001775468,0.006955544,0.004013717,0.003572714,0.001975494],"category_scores_gemma":[0.5522661,0.001293587,0.002563829,0.01116385,0.004669744,0.009290813,0.004865046,0.004400136,0.001214192],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003507428,"about_ca_system_score_gemma":0.002702758,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003805264,"about_ca_topic_score_gemma":0.002927595,"domain_scores_codex":[0.8319094,0.08507289,0.0109145,0.01958609,0.05027729,0.002239799],"domain_scores_gemma":[0.3363208,0.5252619,0.02699884,0.08046687,0.02997361,0.0009778818],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001197008,0.0002555746,0.09124732,0.002021319,0.002992732,0.0003564498,0.00211375,0.2885578,0.00375896,0.1562261,0.01372923,0.4375438],"study_design_scores_gemma":[0.0001535351,0.0006578689,0.04053127,0.0007153678,0.0006785584,0.0009074734,0.0003244592,0.6826802,0.01284773,0.2501026,0.01001958,0.0003814411],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04577175,0.006599812,0.9335715,0.000757129,0.0003260046,0.0002645917,0.001223097,0.00292724,0.008558953],"genre_scores_gemma":[0.8223702,0.00139341,0.169208,0.0006164747,0.0007398521,0.0008964523,0.00236134,0.001317466,0.00109674],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1368204,"threshold_uncertainty_score":0.7235842,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2106377293","doi":"10.1109/hicss.1999.772650","title":"The functionality attribute of cybergenres","year":2003,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":106,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Tuple; The Internet; Class (philosophy); Content (measure theory); World Wide Web; Multimedia; Information retrieval; Artificial intelligence; Mathematics","authors":[{"name":"Michael Shepherd","is_ca":true},{"name":"Carolyn Watters","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01553119694393844,"gpt":0.2669524419538304,"spread":0.251421245009892,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001608356,0.0004720726,0.0002274668,0.004067788,0.001732007,0.004369064,0.0005110772,0.0008481338,0.005102669],"category_scores_gemma":[0.012051,0.0001933119,0.0004121122,0.002163846,0.002904566,0.006150357,0.00181434,0.0009514745,0.0009889402],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001131849,"about_ca_system_score_gemma":0.0005555176,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001290819,"about_ca_topic_score_gemma":0.001110271,"domain_scores_codex":[0.9975048,0.0007095598,0.0003335805,0.0002624743,0.0009901426,0.0001994493],"domain_scores_gemma":[0.9889469,0.004240253,0.001499074,0.001997402,0.002486445,0.0008299607],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003266793,0.0001176465,0.04778701,0.0006618139,0.0000444267,0.001999137,0.04065489,0.00129414,0.02600791,0.6084034,0.009414103,0.2632888],"study_design_scores_gemma":[0.00004654462,0.0004221931,0.1395198,0.00068915,0.0001270709,0.01394617,0.02586781,0.01195535,0.01701743,0.2470403,0.5431498,0.0002183329],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5545015,0.001406215,0.07748684,0.002227028,0.0005608096,0.0002073423,0.001190517,0.00124113,0.3611786],"genre_scores_gemma":[0.967967,0.0004612145,0.01372517,0.0001865214,0.0002672163,0.00006650353,0.0007660494,0.0002175401,0.01634282],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005102669,"threshold_uncertainty_score":0.01707011,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2250595592","doi":"10.3115/v1/w14-4407","title":"A Template-based Abstractive Meeting Summarization: Leveraging Summary and Source Text Relationships","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":100,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Automatic summarization; Readability; Computer science; Template; Sentence; Natural language processing; Information retrieval; Artificial intelligence; Programming language","authors":[{"name":"Tatsuro Oya","is_ca":true},{"name":"Yashar Mehdad","is_ca":true},{"name":"Giuseppe Carenini","is_ca":true},{"name":"Raymond T. Ng","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02111276075879438,"gpt":0.2450902321805714,"spread":0.223977471421777,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002117283,0.00182808,0.0009840046,0.003078339,0.0008085145,0.002457667,0.001708153,0.001274474,0.005993003],"category_scores_gemma":[0.009471575,0.0005744559,0.0009821591,0.001729978,0.0004095003,0.002700882,0.001333699,0.001304942,0.007485603],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003801481,"about_ca_system_score_gemma":0.0009012424,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00159174,"about_ca_topic_score_gemma":0.001933287,"domain_scores_codex":[0.9974836,0.0007109686,0.0003344828,0.000733846,0.0006602872,0.00007688653],"domain_scores_gemma":[0.9941604,0.001923318,0.00092092,0.0008545871,0.001921715,0.0002190644],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004521428,0.0001514327,0.001335132,0.001249127,0.0001924115,0.0003254726,0.001098783,0.006429773,0.1361323,0.002380727,0.02386338,0.8263894],"study_design_scores_gemma":[0.000252088,0.001407461,0.009510899,0.0003268991,0.001012712,0.001918121,0.001457165,0.4429461,0.3657434,0.01034988,0.1646103,0.0004650438],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0130277,0.0008559242,0.9563157,0.0003780963,0.0002942896,0.0004091952,0.002639249,0.0239826,0.002097313],"genre_scores_gemma":[0.06167391,0.0004843033,0.9239806,0.0001353559,0.0003009682,0.0002989705,0.0085094,0.0009308405,0.003685714],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005993003,"threshold_uncertainty_score":0.02004862,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2962704246","doi":"10.18653/v1/p18-1062","title":"Unsupervised Abstractive Meeting Summarization with Multi-Sentence Compression and Budgeted Submodular Maximization","year":2018,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":100,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Automatic summarization; Computer science; Zhàng; Submodular set function; Natural language processing; Sentence; Maximization; Artificial intelligence; Linguistics; Volume (thermodynamics); Mathematics; Philosophy; History; Combinatorics","authors":[{"name":"Guokan Shang","is_ca":true},{"name":"Wensi Ding","is_ca":true},{"name":"Zekun Zhang","is_ca":true},{"name":"Antoine J.‐P. Tixier","is_ca":false},{"name":"Polykarpos Meladianos","is_ca":true},{"name":"Michalis Vazirgiannis","is_ca":true},{"name":"Jean-Pierre Lorré","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01322545130685726,"gpt":0.2521054262172721,"spread":0.2388799749104148,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001854928,0.002605617,0.002758964,0.002130936,0.000924187,0.001771879,0.002503692,0.001702582,0.005144968],"category_scores_gemma":[0.004487707,0.0008437551,0.00149185,0.002816517,0.0005595272,0.003273378,0.002014599,0.001978604,0.004284915],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008058252,"about_ca_system_score_gemma":0.001625727,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003893706,"about_ca_topic_score_gemma":0.009568098,"domain_scores_codex":[0.9984605,0.0005387361,0.000134896,0.0004383372,0.0002704689,0.0001571615],"domain_scores_gemma":[0.9978661,0.001065987,0.0001861205,0.0003053967,0.0004700793,0.0001062924],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006827464,0.0003312799,0.0008670987,0.0006275889,0.000314798,0.0002849404,0.0004253329,0.08520941,0.01665597,0.006515551,0.04947925,0.8386061],"study_design_scores_gemma":[0.00008679544,0.0001914004,0.0004950899,0.0000364451,0.0001477735,0.0000993208,0.0002223799,0.9705153,0.006748754,0.01413459,0.007290458,0.00003173182],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01375757,0.00248723,0.973183,0.0007266427,0.0003385549,0.0002395913,0.001916883,0.005099972,0.002250447],"genre_scores_gemma":[0.2131144,0.001193421,0.7504082,0.0005240426,0.001211996,0.000729747,0.01880593,0.001265491,0.01274672],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005144968,"threshold_uncertainty_score":0.01721156,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2950657507","doi":"10.1073/pnas.1914370116","title":"Predicting research trends with semantic and neural networks with an application in quantum physics","year":2020,"lang":"en","type":"article","venue":"Proceedings of the National Academy of Sciences","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":97,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Vector Institute; University of Toronto","funders":"Universität Wien; Austrian Science Fund","keywords":"Artificial neural network; Quantum; Computer science; Physics; Artificial intelligence; Quantum mechanics","authors":[{"name":"Mario Krenn","is_ca":true},{"name":"Anton Zeilinger","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06736840998995927,"gpt":0.3578810530827379,"spread":0.2905126430927786,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.00214705,0.0006888016,0.0004289616,0.005683281,0.0008631969,0.00130792,0.0005830762,0.001262892,0.001286381],"category_scores_gemma":[0.01137012,0.0003054651,0.0008408325,0.00485237,0.0008777213,0.002627171,0.001000799,0.001129495,0.0002413266],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001403323,"about_ca_system_score_gemma":0.0007833279,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006656274,"about_ca_topic_score_gemma":0.01179088,"domain_scores_codex":[0.9993187,0.0002820104,0.00004920054,0.0001745545,0.0001213477,0.00005409624],"domain_scores_gemma":[0.9949558,0.003398678,0.0006987351,0.0002916085,0.0005157142,0.0001393989],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005072645,0.0005315731,0.05997109,0.0004451922,0.0004505233,0.0003791533,0.0005083529,0.6489235,0.003511566,0.05257102,0.007735908,0.2244649],"study_design_scores_gemma":[0.00001754846,0.00002852801,0.003892693,0.00003071993,0.00003805057,0.00003702415,0.00005817161,0.9409529,0.0005281889,0.05291215,0.001488681,0.00001543119],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4356224,0.003754771,0.5378304,0.007410042,0.0003645582,0.0001804834,0.004554681,0.00161562,0.008667121],"genre_scores_gemma":[0.8386218,0.001214232,0.1548229,0.0003959714,0.0002563768,0.000183841,0.002946889,0.0000663207,0.001491815],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9943167,"threshold_uncertainty_score":0.01323503,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2953722276","doi":"10.1016/j.ipm.2019.102063","title":"A multi-centrality index for graph-based keyword extraction","year":2019,"lang":"en","type":"article","venue":"Information Processing & Management","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":92,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"Fundação de Amparo à Pesquisa do Estado de São Paulo","keywords":"Centrality; Betweenness centrality; PageRank; Computer science; Clustering coefficient; Cluster analysis; Graph; Artificial intelligence; Data mining; Natural language processing; Information retrieval; Theoretical computer science; Mathematics; Statistics","authors":[{"name":"Didier A. Vega‐Oliveros","is_ca":false},{"name":"Pedro Spoljaric Gomes","is_ca":false},{"name":"Evangelos Milios","is_ca":true},{"name":"Lilian Berton","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01397295543473869,"gpt":0.2989504196007305,"spread":0.2849774641659918,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000880022,0.000815603,0.001329912,0.01497734,0.001379768,0.002121515,0.001229353,0.0009950873,0.00347837],"category_scores_gemma":[0.006668062,0.0003532494,0.0009323814,0.01215498,0.0003893241,0.002796277,0.001346748,0.0006711809,0.00258424],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009542897,"about_ca_system_score_gemma":0.001702928,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0054911,"about_ca_topic_score_gemma":0.01145477,"domain_scores_codex":[0.9982261,0.0002054589,0.0002076844,0.0003482548,0.0008738437,0.0001387304],"domain_scores_gemma":[0.9964018,0.001246501,0.0003928252,0.000335048,0.001412188,0.0002117262],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008157961,0.0004269304,0.01381026,0.001131252,0.0004602203,0.0005727409,0.0004077078,0.0258945,0.08161053,0.01803473,0.03562684,0.8212084],"study_design_scores_gemma":[0.0001431238,0.0003619132,0.01462856,0.0001478143,0.000450775,0.001778369,0.000434206,0.8581312,0.04307064,0.04038145,0.04026389,0.0002082036],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06871323,0.0028866,0.9058158,0.0003862192,0.0003432556,0.0004578163,0.009036384,0.007059675,0.005301059],"genre_scores_gemma":[0.3318658,0.001166846,0.6457255,0.0001296729,0.0003708132,0.0004263093,0.01443012,0.0006524317,0.005232514],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01497734,"threshold_uncertainty_score":0.01163632,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2150630862","doi":"10.1075/ml.5.1.06baa","title":"A real experiment is a factorial experiment?","year":2010,"lang":"en","type":"article","venue":"The Mental Lexicon","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":89,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Factorial experiment; Mathematics; Statistics; Computer science; Artificial intelligence","authors":[{"name":"R. Harald Baayen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01636144710900018,"gpt":0.3160083406347708,"spread":0.2996468935257707,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2238169,0.002140238,0.004014705,0.001388257,0.002606492,0.0079155,0.00327262,0.007083857,0.01908097],"category_scores_gemma":[0.456216,0.001778264,0.002785393,0.002519093,0.01749876,0.01875325,0.004050239,0.005525784,0.002980238],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003968998,"about_ca_system_score_gemma":0.003281313,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008887682,"about_ca_topic_score_gemma":0.0007485396,"domain_scores_codex":[0.7563086,0.1863325,0.01063635,0.0246128,0.02036039,0.001749403],"domain_scores_gemma":[0.3731064,0.5203212,0.02623819,0.0657061,0.01207588,0.002552252],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01774105,0.002958094,0.01330922,0.01385679,0.004014832,0.0003424742,0.0107414,0.003340271,0.008836301,0.6502941,0.04975337,0.2248121],"study_design_scores_gemma":[0.00918514,0.01679409,0.01817054,0.004189906,0.002358572,0.0005973397,0.002781973,0.01030845,0.00671782,0.6411614,0.2867434,0.0009913284],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08040334,0.01575153,0.7128447,0.05739011,0.02612697,0.01664115,0.003117468,0.003235008,0.08448962],"genre_scores_gemma":[0.3834527,0.004711737,0.5128312,0.02935338,0.007038816,0.05521366,0.001018642,0.0008627363,0.005517154],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2238169,"threshold_uncertainty_score":0.9571719,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2107363066","doi":"10.1002/asi.23367","title":"The invariant distribution of references in scientific articles","year":2015,"lang":"en","type":"article","venue":"Journal of the Association for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"Canada Research Chairs","keywords":"Bibliometrics; Computer science; Scientific communication; Section (typography); Information retrieval; Library science; Data science","authors":[{"name":"Marc Bertin","is_ca":true},{"name":"Iana Atanassova","is_ca":true},{"name":"Yves Gingras","is_ca":true},{"name":"Vincent Larivière","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01949208080998927,"gpt":0.2780394076808985,"spread":0.2585473268709092,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.009893283,0.0001966289,0.0006025152,0.01462529,0.000965548,0.00517391,0.0008142925,0.0008326152,0.00493458],"category_scores_gemma":[0.1255065,0.0002989279,0.0003894501,0.01802154,0.002907359,0.006127981,0.002391184,0.0007844868,0.002033937],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001327081,"about_ca_system_score_gemma":0.0008181967,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001466676,"about_ca_topic_score_gemma":0.001093078,"domain_scores_codex":[0.9898689,0.002508001,0.001669281,0.002468429,0.002897431,0.0005880104],"domain_scores_gemma":[0.8588831,0.07362419,0.03389973,0.01455557,0.01735485,0.001682481],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007818285,0.0001298499,0.6860188,0.0007601377,0.0004714213,0.0007531086,0.01519226,0.002503662,0.01213918,0.0603211,0.004328539,0.2166002],"study_design_scores_gemma":[0.00004576252,0.0002656889,0.8891271,0.0002490481,0.0002353169,0.001167356,0.006177196,0.004802176,0.005391573,0.07247133,0.01992394,0.0001433789],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9504974,0.004282826,0.01909122,0.0009115088,0.0001966726,0.00007532012,0.002236942,0.0003249504,0.02238308],"genre_scores_gemma":[0.9939728,0.0007696851,0.002460523,0.00007877742,0.000191186,0.00004083108,0.0008817585,0.00007744744,0.001526921],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9853747,"threshold_uncertainty_score":0.05232131,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2089286873","doi":"10.1037/h0087425","title":"The p-value fallacy and how to avoid it.","year":2003,"lang":"en","type":"article","venue":"Canadian Journal of Experimental Psychology/Revue canadienne de psychologie expérimentale","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Fallacy; Null hypothesis; Psychology; Value (mathematics); Statistical hypothesis testing; Interpretation (philosophy); p-value; Econometrics; Statistics; Alternative hypothesis; Inference; Cognitive psychology; Mathematics; Epistemology; Artificial intelligence; Computer science","authors":[{"name":"Peter Dixon","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0283444507547641,"gpt":0.3147546726853978,"spread":0.2864102219306337,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1510031,0.003263132,0.003460615,0.007539434,0.003328864,0.007740777,0.007008536,0.01444783,0.004870548],"category_scores_gemma":[0.5519866,0.002130364,0.00179871,0.007549563,0.03707583,0.01840517,0.007878626,0.02604416,0.005877331],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003512912,"about_ca_system_score_gemma":0.005748644,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002181739,"about_ca_topic_score_gemma":0.001964088,"domain_scores_codex":[0.7872633,0.1545486,0.01196285,0.01213875,0.03325267,0.0008337629],"domain_scores_gemma":[0.4606815,0.489124,0.01219681,0.01869696,0.01758478,0.001715917],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00031019,0.0001252017,0.002966444,0.002685368,0.0006386502,0.001232304,0.002649418,0.002012539,0.0008420717,0.531331,0.198807,0.2563998],"study_design_scores_gemma":[0.00009579043,0.00006477656,0.0005395723,0.001206663,0.00007654163,0.001009819,0.0003910792,0.003701366,0.0008594501,0.9067917,0.08512846,0.0001347122],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001896184,0.03488715,0.7837365,0.1475646,0.01544033,0.0005282708,0.0003468427,0.001944755,0.01365548],"genre_scores_gemma":[0.06213991,0.01814787,0.8131175,0.07936949,0.01527366,0.003725013,0.0002446592,0.001695026,0.006286833],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8489969,"threshold_uncertainty_score":0.7985902,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2153207410","doi":"10.1145/1978942.1979167","title":"Review spotlight","year":2011,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":83,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Product (mathematics); Adjective; Quality (philosophy); World Wide Web; Noun; Information retrieval; Natural language processing","authors":[{"name":"Koji Yatani","is_ca":true},{"name":"Michael Novati","is_ca":true},{"name":"Andrew Trusty","is_ca":true},{"name":"Khai N. Truong","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04088631501555477,"gpt":0.2811978908767365,"spread":0.2403115758611817,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003507061,0.001281104,0.001442718,0.004890063,0.000906232,0.002550552,0.001786377,0.0009771112,0.1188753],"category_scores_gemma":[0.02036661,0.0008007372,0.000752456,0.001946598,0.0003014027,0.0029128,0.002008267,0.0009196685,0.07234846],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004718191,"about_ca_system_score_gemma":0.001159055,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006226579,"about_ca_topic_score_gemma":0.001397499,"domain_scores_codex":[0.9979222,0.0005485934,0.0002351662,0.0004366726,0.0007370925,0.0001202381],"domain_scores_gemma":[0.9713597,0.009660826,0.002030408,0.004009534,0.0106225,0.002317139],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001346324,0.00009174262,0.001596427,0.00207585,0.00009371791,0.0003830759,0.0005440819,0.0001801483,0.01468663,0.001805121,0.6060801,0.3711168],"study_design_scores_gemma":[0.0002905457,0.0004456285,0.004835257,0.0002954306,0.0001197396,0.001088799,0.0002553322,0.004926472,0.02286951,0.002599622,0.9621205,0.0001531414],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"software","genre_gemma":"other","genre_scores_codex":[0.03357842,0.0129368,0.3516715,0.00385199,0.00623992,0.008969997,0.03848661,0.3794043,0.1648605],"genre_scores_gemma":[0.1662072,0.006153611,0.4056402,0.004242068,0.00553324,0.008022251,0.04869043,0.04201855,0.3134924],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.8811247,"threshold_uncertainty_score":0.3976776,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2032328503","doi":"10.1007/s10791-009-9108-x","title":"Document clustering of scientific texts using citation contexts","year":2009,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Cluster analysis; Information retrieval; Computer science; Document clustering; Citation; Vocabulary; Context (archaeology); Similarity (geometry); Representation (politics); Document retrieval; Natural language processing; Artificial intelligence; World Wide Web; Linguistics","authors":[{"name":"Bader Aljaber","is_ca":false},{"name":"Nicola Stokes","is_ca":false},{"name":"James Bailey","is_ca":false},{"name":"Jian Pei","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0160256859248469,"gpt":0.2950436954188471,"spread":0.2790180094940002,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.001669988,0.0008086043,0.001106689,0.02648126,0.002412321,0.004088819,0.001049552,0.001127128,0.003157671],"category_scores_gemma":[0.01184732,0.0003891796,0.001247355,0.02233986,0.0006420256,0.002487009,0.00120861,0.0009756474,0.002097595],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001043688,"about_ca_system_score_gemma":0.002374801,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00395782,"about_ca_topic_score_gemma":0.008434736,"domain_scores_codex":[0.9978617,0.0005519332,0.0003084471,0.0004282437,0.0006915503,0.0001581303],"domain_scores_gemma":[0.9923053,0.003579697,0.0006003455,0.0005478903,0.002680715,0.0002859683],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008256128,0.0004109048,0.01502229,0.001657673,0.0004256607,0.0003710898,0.001342308,0.01013203,0.03755221,0.01368703,0.01772108,0.900852],"study_design_scores_gemma":[0.000441689,0.001151652,0.100115,0.001224836,0.003434505,0.002277089,0.003867063,0.507206,0.1036586,0.1263126,0.1497516,0.0005594292],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4352847,0.02988485,0.488127,0.002138877,0.00194183,0.001189411,0.009118278,0.005384562,0.02693051],"genre_scores_gemma":[0.6016658,0.006460908,0.3663079,0.0001524675,0.001650409,0.0005931752,0.01205742,0.0006688847,0.01044304],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9735187,"threshold_uncertainty_score":0.01056349,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4255165398","doi":"10.4018/978-1-59140-441-5.ch008","title":"KEA","year":2004,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; Information retrieval; Metadata; Feature (linguistics); Artificial intelligence; License; Natural language processing; World Wide Web; Linguistics","authors":[{"name":"Ian H. Witten","is_ca":false},{"name":"Gordon W. Paynter","is_ca":false},{"name":"Eibe Frank","is_ca":false},{"name":"Carl Gutwin","is_ca":true},{"name":"Craig G. Nevill-Manning","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01472894233422973,"gpt":0.2612281464799091,"spread":0.2464992041456794,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004770194,0.001210922,0.0006368831,0.0035319,0.0008218603,0.004018272,0.001320837,0.001123618,0.1955127],"category_scores_gemma":[0.002170091,0.0005691048,0.0006532796,0.003358457,0.0004835987,0.006292806,0.001540357,0.001405963,0.2323909],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009371066,"about_ca_system_score_gemma":0.0008497924,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009194372,"about_ca_topic_score_gemma":0.001967882,"domain_scores_codex":[0.9995723,0.00004527823,0.00003528404,0.0001146047,0.0001990647,0.00003333758],"domain_scores_gemma":[0.9989719,0.0003174786,0.00004559111,0.000264668,0.0003160039,0.00008430014],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00009068071,0.00004375,0.0002798663,0.0007653131,0.00001749714,0.0001203397,0.0003720821,0.0005041347,0.006673384,0.0604321,0.3464848,0.5842161],"study_design_scores_gemma":[0.000004255417,0.000008106397,0.0002113021,0.00006583202,0.000005091426,0.0002632844,0.00004042081,0.0005522855,0.001515178,0.006155426,0.9911696,0.000009309269],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"methods","genre_scores_codex":[0.003851584,0.01094228,0.228143,0.001664751,0.001263021,0.0003405833,0.0178122,0.05447026,0.6815124],"genre_scores_gemma":[0.01781695,0.007363598,0.1493434,0.0008846979,0.0003416782,0.0002392157,0.02279421,0.01104946,0.7901669],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.1955127,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2950982165","doi":"","title":"Coherent Keyphrase Extraction via Web Mining","year":2003,"lang":"en","type":"preprint","venue":"NPARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Information retrieval; Task (project management); Cluster analysis; Search engine indexing; Natural language processing; Artificial intelligence","authors":[{"name":"Peter D. Turney","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02164678121118828,"gpt":0.3037006518615237,"spread":0.2820538706503354,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001854458,0.001757103,0.001325236,0.01169834,0.001288181,0.003326512,0.002126666,0.001529108,0.00597153],"category_scores_gemma":[0.01141131,0.0009180094,0.001944947,0.009417628,0.0007931869,0.005210042,0.002621494,0.001507844,0.00962709],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008847492,"about_ca_system_score_gemma":0.002123325,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00290203,"about_ca_topic_score_gemma":0.004575515,"domain_scores_codex":[0.9970384,0.0004223253,0.0003738699,0.0009199655,0.001016067,0.0002293082],"domain_scores_gemma":[0.994712,0.001847965,0.000599151,0.001167812,0.001508159,0.0001648586],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003318055,0.0002762333,0.003355279,0.0009397077,0.0001597118,0.0007349236,0.0004388819,0.006327135,0.04020915,0.009238003,0.02101937,0.9169698],"study_design_scores_gemma":[0.0002199072,0.0003392743,0.007722676,0.0003499089,0.0003281633,0.002740883,0.001393315,0.5796871,0.1502817,0.09044608,0.1662429,0.0002481805],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02293778,0.001704233,0.9441935,0.0004324813,0.0001625078,0.0006867388,0.00344832,0.02171515,0.004719351],"genre_scores_gemma":[0.09711329,0.001034021,0.8848361,0.0001930408,0.0001350487,0.000315903,0.01026113,0.0009857817,0.005125626],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01169834,"threshold_uncertainty_score":0.01997674,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2100784475","doi":"10.1177/1461445612466468","title":"Rhetorical relations in multimodal documents","year":2013,"lang":"en","type":"article","venue":"Discourse Studies","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Rhetorical question; Presentational and representational acting; Computer science; Natural language processing; Coherence (philosophical gambling strategy); Linguistics; Categorization; Set (abstract data type); Artificial intelligence; Subject (documents); Mathematics; World Wide Web; Philosophy","authors":[{"name":"Maite Taboada","is_ca":true},{"name":"Christopher Habel","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0271779675503826,"gpt":0.3780032763737782,"spread":0.3508253088233956,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003057532,0.0002932462,0.0004401443,0.008957287,0.003328199,0.004170448,0.0005977471,0.0008709971,0.003928193],"category_scores_gemma":[0.02695549,0.0002470145,0.0002216384,0.01025542,0.003454553,0.005720222,0.001956616,0.001298423,0.0003817231],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001968555,"about_ca_system_score_gemma":0.0009456119,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003178165,"about_ca_topic_score_gemma":0.004703914,"domain_scores_codex":[0.9968194,0.001850401,0.0001978813,0.0003643926,0.000650685,0.0001172072],"domain_scores_gemma":[0.9646251,0.02930022,0.002567416,0.001253811,0.001956612,0.0002968608],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0005904482,0.0003921943,0.04019371,0.003357911,0.0001147471,0.003519864,0.390907,0.006334382,0.02817662,0.2485247,0.00919554,0.2686928],"study_design_scores_gemma":[0.0001832148,0.0004455663,0.1645854,0.001849331,0.0002877434,0.003870846,0.1984275,0.04363082,0.0374678,0.1681557,0.3807268,0.0003692997],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9169292,0.004673783,0.0495817,0.001582393,0.00008736658,0.0001840383,0.002147302,0.0002652579,0.02454893],"genre_scores_gemma":[0.9757976,0.0006843722,0.02039202,0.00006422755,0.00006053331,0.0001876341,0.0009142943,0.00008585395,0.001813543],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008957287,"threshold_uncertainty_score":0.01616991,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2161427841","doi":"","title":"Summarizing Emails with Conversational Cohesion and Subjectivity","year":2008,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Automatic summarization; Cohesion (chemistry); Computer science; PageRank; Cosine similarity; Natural language processing; Artificial intelligence; Graph; Sentence; Empirical research; Information retrieval; Subjectivity; Theoretical computer science; Pattern recognition (psychology); Mathematics","authors":[{"name":"Giuseppe Carenini","is_ca":true},{"name":"Raymond T. Ng","is_ca":true},{"name":"Xiaodong Zhou","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01336756495223629,"gpt":0.2436825173727591,"spread":0.2303149524205229,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003324555,0.001226314,0.0009969994,0.004025226,0.0008412492,0.002490154,0.0007553663,0.001119899,0.0009375978],"category_scores_gemma":[0.02397137,0.0005320183,0.0007730604,0.002409586,0.0006215124,0.005029156,0.001353118,0.0006785095,0.0005331694],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005924078,"about_ca_system_score_gemma":0.0006183968,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001079974,"about_ca_topic_score_gemma":0.001503333,"domain_scores_codex":[0.9968136,0.001583424,0.0003205725,0.0005923277,0.0005747551,0.0001153062],"domain_scores_gemma":[0.9846705,0.01024063,0.00188392,0.001042944,0.001932597,0.0002294712],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009966069,0.0002481605,0.01733918,0.002727068,0.0006065104,0.0008823901,0.006635359,0.06530467,0.04592604,0.02422825,0.007312474,0.8277934],"study_design_scores_gemma":[0.00009275175,0.0007253079,0.0205816,0.0003130655,0.00109194,0.0009415674,0.004252906,0.7659903,0.04293973,0.1387453,0.02412135,0.0002041328],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1160832,0.002279369,0.8766132,0.0006805406,0.000121476,0.0002138822,0.0004842319,0.001350205,0.002173785],"genre_scores_gemma":[0.5672662,0.001268927,0.4265423,0.0001388247,0.0004897352,0.0002102353,0.001941613,0.0002504736,0.001891817],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004025226,"threshold_uncertainty_score":0.01758212,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W152717974","doi":"10.63317/5h9g23tpswhr","title":"Automatically Identifying Changes in the Semantic Orientation of Words","year":2010,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Task (project management); Natural language processing; Word (group theory); Artificial intelligence; Orientation (vector space); Semantic change; Meaning (existential); Baseline (sea); Word-sense disambiguation; Semantic role labeling; Linguistics; Psychology; Mathematics","authors":[{"name":"Paul Cook","is_ca":true},{"name":"Suzanne Stevenson","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01640136917437291,"gpt":0.3180775501979433,"spread":0.3016761810235704,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008762213,0.0007564793,0.0005716384,0.004839412,0.0005432747,0.00206947,0.0006447171,0.0008537995,0.001469341],"category_scores_gemma":[0.006003547,0.0003947956,0.0006233265,0.003224559,0.0007370632,0.003070175,0.001172648,0.000944791,0.001754612],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008255879,"about_ca_system_score_gemma":0.0008612178,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004963121,"about_ca_topic_score_gemma":0.008348058,"domain_scores_codex":[0.998576,0.0001821846,0.0001897791,0.0006586268,0.0002680679,0.0001253375],"domain_scores_gemma":[0.9966137,0.001057437,0.0007312804,0.0003810396,0.001054091,0.0001625436],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008896433,0.0003635176,0.1565465,0.0007395264,0.0001768419,0.0006237101,0.002354795,0.004883505,0.1020802,0.002802447,0.01424713,0.7142922],"study_design_scores_gemma":[0.0003045464,0.0009077799,0.4850457,0.0003193175,0.0005319682,0.002616457,0.01075184,0.2722818,0.1138631,0.03113211,0.08188752,0.0003578445],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9037272,0.001606279,0.07369514,0.0004438131,0.0002890934,0.0002709474,0.007162301,0.004867453,0.007937814],"genre_scores_gemma":[0.8915421,0.000483458,0.09284019,0.0001525834,0.0001111504,0.0001681794,0.01184586,0.000353565,0.002502921],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004963121,"threshold_uncertainty_score":0.009868503,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2138954094","doi":"10.1145/1097047.1097059","title":"Narrative text classification for automatic key phrase extraction in web document corpora","year":2005,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Automatic summarization; Information retrieval; Plain text; Key (lock); Natural language processing; Phrase; tf–idf; Artificial intelligence; Ranking (information retrieval); Document clustering; Information extraction; HTML element; Web page; Keyword extraction; Cluster analysis; World Wide Web; Term (time)","authors":[{"name":"Yongzheng Zhang","is_ca":true},{"name":"A. Nur Zincir‐Heywood","is_ca":true},{"name":"Evangelos Milios","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02461358341141394,"gpt":0.3336148891383208,"spread":0.3090013057269069,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004971606,0.001279869,0.001024761,0.007092583,0.001309666,0.001990597,0.001043538,0.0009905315,0.006275678],"category_scores_gemma":[0.02771288,0.0004659655,0.0009978147,0.005589753,0.0005008634,0.004242604,0.0009427947,0.0009187756,0.006822853],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007999136,"about_ca_system_score_gemma":0.001403273,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002603024,"about_ca_topic_score_gemma":0.003520015,"domain_scores_codex":[0.9961916,0.001404987,0.0008353284,0.0005765585,0.0008472668,0.0001441361],"domain_scores_gemma":[0.9856133,0.008096746,0.001078833,0.00126459,0.003703939,0.0002426057],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007460054,0.0002627656,0.005378508,0.002387255,0.0001521974,0.0003569991,0.001078642,0.004414628,0.1200738,0.003729195,0.02358152,0.8378384],"study_design_scores_gemma":[0.0006249656,0.001603589,0.04594003,0.0006633648,0.0005681191,0.002670287,0.002366386,0.5175076,0.2818992,0.01136628,0.1344045,0.0003857906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1040356,0.002914625,0.8533756,0.0006291355,0.0003279838,0.001950045,0.01033765,0.02095034,0.00547905],"genre_scores_gemma":[0.09328279,0.0006168459,0.8852324,0.00007071499,0.0001240971,0.00138948,0.01720149,0.0005766425,0.001505452],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007092583,"threshold_uncertainty_score":0.02629268,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2603222250","doi":"10.1017/s0269888917000029","title":"The state of the art in semantic relatedness: a framework for comparison","year":2017,"lang":"en","type":"article","venue":"The Knowledge Engineering Review","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; WordNet; Closeness; Semantic similarity; Similarity (geometry); Strengths and weaknesses; Representation (politics); Domain (mathematical analysis); Metric (unit); Data science; Information retrieval; Artificial intelligence; Mathematics","authors":[{"name":"Yue Feng","is_ca":true},{"name":"Ebrahim Bagheri","is_ca":true},{"name":"Faezeh Ensan","is_ca":false},{"name":"Jelena Jovanović","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02277658548795489,"gpt":0.3350866112312231,"spread":0.3123100257432682,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03425916,0.001415523,0.002555996,0.04748476,0.001659429,0.008674193,0.003871921,0.002807056,0.003832999],"category_scores_gemma":[0.08219115,0.0004944279,0.00168076,0.03777338,0.005894435,0.02175762,0.003917056,0.003410329,0.001324956],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006035258,"about_ca_system_score_gemma":0.003710482,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003357406,"about_ca_topic_score_gemma":0.002051172,"domain_scores_codex":[0.9667661,0.01989622,0.002840192,0.002547719,0.007410995,0.000538854],"domain_scores_gemma":[0.9177903,0.06216825,0.004158395,0.004887352,0.01026488,0.0007307557],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00008743895,0.0001317548,0.002697136,0.006822943,0.0003636873,0.00009339768,0.00142026,0.004601386,0.0009317754,0.4964195,0.01509791,0.4713328],"study_design_scores_gemma":[0.00004023686,0.0003522299,0.008228969,0.01467368,0.0004978082,0.0005381702,0.004234087,0.04556042,0.002310549,0.6294649,0.2938845,0.0002145198],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"review","genre_gemma":"methods","genre_scores_codex":[0.00892645,0.5600542,0.3878932,0.01193947,0.001207822,0.0003692281,0.0006989847,0.0004414954,0.02846904],"genre_scores_gemma":[0.2585809,0.2513789,0.4767467,0.003323188,0.003124244,0.001641571,0.001985027,0.0003668758,0.002852573],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.04748476,"threshold_uncertainty_score":0.1811819,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2169142063","doi":"10.1613/jair.3940","title":"Topic Segmentation and Labeling in Asynchronous Conversations","year":2013,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; University of British Columbia","keywords":"Computer science; Asynchronous communication; Segmentation; Artificial intelligence; Conversation; Natural language processing; Exploit; Graph; Linguistics; Theoretical computer science","authors":[{"name":"Shafiq Joty","is_ca":false},{"name":"Giuseppe Carenini","is_ca":true},{"name":"Raymond T. Ng","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1082659095785678,"gpt":0.424850168063926,"spread":0.3165842584853582,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005408508,0.001015378,0.001034947,0.002402256,0.00209882,0.002590633,0.001632388,0.001766325,0.001818956],"category_scores_gemma":[0.02927748,0.0006232549,0.0009398351,0.001832323,0.00122094,0.003867432,0.002327436,0.001740273,0.001550519],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001229982,"about_ca_system_score_gemma":0.00148746,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005213632,"about_ca_topic_score_gemma":0.007229337,"domain_scores_codex":[0.9918485,0.004537428,0.0003562074,0.002092314,0.0008401446,0.0003254047],"domain_scores_gemma":[0.9691306,0.02260645,0.001983828,0.002948555,0.002620153,0.0007104086],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004932778,0.0007007124,0.05377226,0.001993169,0.0004754022,0.001461678,0.02275943,0.1641653,0.09098533,0.05478608,0.02690663,0.5770612],"study_design_scores_gemma":[0.00009919636,0.0001399109,0.01267889,0.00009963029,0.0001345451,0.0003394267,0.001691377,0.8910539,0.0237433,0.05291623,0.01697447,0.0001292227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2004868,0.0009937906,0.7877063,0.0006492131,0.0001770947,0.0003308395,0.001371279,0.002780067,0.005504643],"genre_scores_gemma":[0.7780083,0.0003759026,0.2129153,0.000187475,0.0002595436,0.0005252056,0.003593986,0.0005529955,0.003581311],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005408508,"threshold_uncertainty_score":0.02860326,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2621499279","doi":"10.1037/xlm0000455","title":"Exploring the self-ownership effect: Separating stimulus and response biases.","year":2017,"lang":"en","type":"article","venue":"Journal of Experimental Psychology Learning Memory and Cognition","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"Economic and Social Research Council","keywords":"Categorization; Cognitive psychology; Stimulus (psychology); PsycINFO; Psychology; Perception; Prioritization; Object (grammar); Information processing; Task (project management); Social psychology; Computer science; Artificial intelligence; Business; Economics; Political science; MEDLINE","authors":[{"name":"Marius Golubickis","is_ca":false},{"name":"Johanna K. Falbén","is_ca":false},{"name":"William A. Cunningham","is_ca":true},{"name":"C. Neil Macrae","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09530499901207543,"gpt":0.3885107134879918,"spread":0.2932057144759164,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005253863,0.0002691492,0.0004485003,0.0006224476,0.0002122143,0.001026427,0.0005682809,0.0006543525,0.003348615],"category_scores_gemma":[0.03759355,0.0002621256,0.0004264566,0.0004290213,0.001278991,0.003061712,0.001657031,0.0007648217,0.0002733425],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005716594,"about_ca_system_score_gemma":0.0004948574,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008237892,"about_ca_topic_score_gemma":0.001277858,"domain_scores_codex":[0.9986046,0.0005294449,0.00009976773,0.0003381657,0.0003604432,0.00006763174],"domain_scores_gemma":[0.967036,0.02540204,0.003832505,0.002576608,0.0006373054,0.0005156344],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.005792278,0.00104727,0.180983,0.00176423,0.0004243163,0.000365339,0.004109586,0.003197799,0.4446036,0.02164857,0.0008071836,0.3352569],"study_design_scores_gemma":[0.0002993298,0.003392223,0.6679156,0.0001791448,0.0004395271,0.001312074,0.001191334,0.1091509,0.1557271,0.05727873,0.002988856,0.00012512],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.94606,0.0004831796,0.04747212,0.0002898183,0.00002613988,0.0002156612,0.0002447232,0.0001192284,0.005089012],"genre_scores_gemma":[0.985678,0.0001295329,0.01318299,0.00009248351,0.00001414141,0.00009411128,0.0001229448,0.00003823323,0.000647508],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005253863,"threshold_uncertainty_score":0.02778542,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3027864066","doi":"10.1145/3377939","title":"Story Forest","year":2020,"lang":"en","type":"article","venue":"ACM Transactions on Knowledge Discovery from Data","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Timeline; Computer science; Event (particle physics); Information retrieval; The Internet; Novelty; World Wide Web; Cluster analysis; Set (abstract data type); News aggregator; Graph; Data science; Artificial intelligence; History","authors":[{"name":"Bang Liu","is_ca":true},{"name":"Fred X. Han","is_ca":true},{"name":"Di Niu","is_ca":true},{"name":"Linglong Kong","is_ca":true},{"name":"Kunfeng Lai","is_ca":false},{"name":"Yu Xu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06664237805374586,"gpt":0.3114345391788544,"spread":0.2447921611251086,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006922385,0.001755382,0.0006985309,0.003107492,0.001331704,0.002605887,0.002297776,0.001302518,0.07456609],"category_scores_gemma":[0.003858663,0.0006255303,0.00185235,0.002570536,0.0003868566,0.004175462,0.002283463,0.001478413,0.03810922],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006778545,"about_ca_system_score_gemma":0.001383188,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004977882,"about_ca_topic_score_gemma":0.01420694,"domain_scores_codex":[0.9991633,0.0001269304,0.00005922194,0.0003169754,0.0002345139,0.00009914594],"domain_scores_gemma":[0.9990376,0.0003215243,0.00006552265,0.0002593464,0.0002348382,0.0000811186],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003507535,0.0001809252,0.002870011,0.0009049827,0.0001305217,0.0004553862,0.000242131,0.01235606,0.002915271,0.02594774,0.4491531,0.5044931],"study_design_scores_gemma":[0.0001353358,0.0001163247,0.00206778,0.0002357142,0.0001019398,0.001017794,0.000442149,0.1716942,0.006384603,0.08241154,0.735324,0.00006871013],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01626917,0.003983403,0.5543593,0.002554548,0.0015691,0.002101169,0.1660955,0.08918893,0.1638788],"genre_scores_gemma":[0.09396937,0.002320024,0.4876962,0.001304079,0.0003954143,0.001380324,0.3339684,0.005827719,0.07313858],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.07456609,"threshold_uncertainty_score":0.2494484,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2966026905","doi":"10.24963/ijcai.2019/712","title":"Unsupervised Neural Aspect Extraction with Sememes","year":2019,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Youth Innovation Promotion Association; National Natural Science Foundation of China; Ant Financial Services Group; Tencent; National Key Research and Development Program of China; Youth Innovation Promotion Association of the Chinese Academy of Sciences","keywords":"Computer science; Artificial intelligence; Natural language processing; Semantics (computer science); Coherence (philosophical gambling strategy); Sentence; Context (archaeology); Word (group theory); Artificial neural network; Linguistics","authors":[{"name":"Ling Luo","is_ca":false},{"name":"Xiang Ao","is_ca":false},{"name":"Yan Song","is_ca":false},{"name":"Jinyao Li","is_ca":false},{"name":"Xiaopeng Yang","is_ca":true},{"name":"Qing He","is_ca":false},{"name":"Dong Yu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.007519555874121481,"gpt":0.2530384046953645,"spread":0.2455188488212431,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003281413,0.0007184703,0.0003824233,0.001056051,0.0002643215,0.000593919,0.0007991231,0.0005981361,0.001493644],"category_scores_gemma":[0.001249151,0.0002952286,0.00077461,0.001115268,0.0003951839,0.001627683,0.0006632702,0.0009578317,0.0005683497],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004039497,"about_ca_system_score_gemma":0.0005325122,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002474953,"about_ca_topic_score_gemma":0.009106853,"domain_scores_codex":[0.9997987,0.000034698,0.00001400469,0.00009046133,0.00003992565,0.00002217036],"domain_scores_gemma":[0.9996468,0.0001459914,0.00004154956,0.00006616101,0.00008446303,0.00001501826],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000204772,0.0001906806,0.004583808,0.0002099317,0.0001841642,0.000376549,0.0002362461,0.05305179,0.03995513,0.01029938,0.007583268,0.8831242],"study_design_scores_gemma":[0.00001646634,0.00007098098,0.002605437,0.00001982578,0.00006068112,0.0001334832,0.00005907267,0.9568457,0.0134671,0.0223947,0.004309482,0.0000171259],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09978336,0.0009820708,0.8885842,0.0003502505,0.0001064539,0.0001543946,0.0007224581,0.004908107,0.004408782],"genre_scores_gemma":[0.6624169,0.0005836791,0.3254244,0.0003279846,0.000110546,0.0001861326,0.003010965,0.0002578006,0.007681612],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002474953,"threshold_uncertainty_score":0.004996717,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4322751637","doi":"10.1108/ejm-07-2020-0542","title":"A comparative study of the predictive power of component-based approaches to structural equation modeling","year":2022,"lang":"en","type":"article","venue":"European Journal of Marketing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Manitoba; McGill University","funders":"","keywords":"Structural equation modeling; Component (thermodynamics); Computer science; Predictive power; Predictive modelling; Econometrics; Path analysis (statistics); Partial least squares regression; Machine learning; Mathematics","authors":[{"name":"Gyeongcheol Cho","is_ca":true},{"name":"Sunmee Kim","is_ca":true},{"name":"Jonathan Lee","is_ca":false},{"name":"Heungsun Hwang","is_ca":true},{"name":"Marko Sarstedt","is_ca":false},{"name":"Christian M. Ringle","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.103315132782719,"gpt":0.2758739453762855,"spread":0.1725588125935665,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07857579,0.002326974,0.001222015,0.008296901,0.0009423129,0.005024462,0.002247658,0.001289612,0.002668804],"category_scores_gemma":[0.2238732,0.0008192221,0.002366486,0.009625831,0.002041361,0.005878329,0.002405307,0.002669698,0.0005580463],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002039235,"about_ca_system_score_gemma":0.003411924,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007593775,"about_ca_topic_score_gemma":0.005746965,"domain_scores_codex":[0.9458847,0.0456011,0.0009353952,0.00195126,0.005261087,0.0003663995],"domain_scores_gemma":[0.6337813,0.3371083,0.005138526,0.008970607,0.01389449,0.001106682],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001503489,0.000865189,0.1782707,0.001549911,0.002780015,0.0001552509,0.006168126,0.1102013,0.0006018198,0.06871952,0.004865156,0.6243195],"study_design_scores_gemma":[0.0002336014,0.001430949,0.05190632,0.001225762,0.0009922272,0.0001778307,0.002481041,0.8762562,0.0009586466,0.06043109,0.003700068,0.0002063103],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5764509,0.008882685,0.3850758,0.004362127,0.0004805218,0.001250344,0.0007071898,0.001983815,0.0208065],"genre_scores_gemma":[0.8947133,0.002610601,0.1003611,0.0002373091,0.0001718146,0.0005138653,0.0006464249,0.0001884209,0.0005571758],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.07857579,"threshold_uncertainty_score":0.4155535,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W423015428","doi":"10.1016/j.joi.2015.05.001","title":"Modelling count response variables in informetric studies: Comparison among count, linear, and lognormal regression models","year":2015,"lang":"en","type":"article","venue":"Journal of Informetrics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University","funders":"Capital Medical University","keywords":"Count data; Statistics; Negative binomial distribution; Poisson regression; Akaike information criterion; Mathematics; Regression analysis; Linear regression; Poisson distribution; Overdispersion; Log-normal distribution; Generalized linear model; Binomial regression; Regression diagnostic; Econometrics; Polynomial regression; Population","authors":[{"name":"Isola Ajiferuke","is_ca":true},{"name":"Felix Famoye","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.103223804081119,"gpt":0.3570945740796192,"spread":0.2538707699985002,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.07055525,0.001770683,0.00243211,0.004490251,0.0008447333,0.005542858,0.005063356,0.003785598,0.005513296],"category_scores_gemma":[0.2788167,0.0008557212,0.003543701,0.006461505,0.002147052,0.008207985,0.002640764,0.003516645,0.001320098],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003072211,"about_ca_system_score_gemma":0.003096512,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009292494,"about_ca_topic_score_gemma":0.007711653,"domain_scores_codex":[0.9397513,0.05364316,0.001621268,0.002237685,0.002118354,0.0006283176],"domain_scores_gemma":[0.454904,0.5239224,0.01007757,0.005106936,0.005247912,0.0007411545],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007497935,0.001512008,0.1572095,0.004940731,0.004181415,0.0008308645,0.005764341,0.4057179,0.001091361,0.1348407,0.004204678,0.2722085],"study_design_scores_gemma":[0.0003264513,0.001086855,0.01134864,0.0006959466,0.001186179,0.0003466572,0.00188746,0.9054806,0.0006569601,0.07434648,0.002496466,0.0001413796],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.27918,0.005025704,0.7071362,0.003002206,0.0003149204,0.000931716,0.001085617,0.0004438759,0.002879741],"genre_scores_gemma":[0.8191155,0.003813606,0.1694294,0.0005021491,0.0002705375,0.00168751,0.001081518,0.000221146,0.003878489],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9955097,"threshold_uncertainty_score":0.3731363,"prediction_status":"machine_predicted_unvalidated"},"labels":[{"model":"gemma","categories":["metaresearch"],"domain":"methods","study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":["bibliometrics"],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W81794110","doi":"","title":"Adaptive User Interfaces for Intelligent E-Learning: Issues and Trends","year":2004,"lang":"en","type":"article","venue":"Journal of the Association for Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McMaster University; University of Waterloo","funders":"","keywords":"Computer science; Human–computer interaction; User interface; The Internet; Context (archaeology); Multimedia; Domain (mathematical analysis); World Wide Web; Process (computing); User interface design; User experience design","authors":[{"name":"Abdul Rahim Ahmad","is_ca":true},{"name":"Otman Basir","is_ca":true},{"name":"Khaled Hassanein","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01541633088549207,"gpt":0.2851988313401981,"spread":0.269782500454706,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00918144,0.0007215497,0.001128705,0.003080385,0.0008259683,0.008028675,0.002471557,0.006195955,0.00541063],"category_scores_gemma":[0.01314254,0.0006115851,0.0004950131,0.006020627,0.003229921,0.0175241,0.00196551,0.005331232,0.002491257],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00181934,"about_ca_system_score_gemma":0.001585942,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001183651,"about_ca_topic_score_gemma":0.001153955,"domain_scores_codex":[0.9962829,0.001081743,0.0004648458,0.0005128504,0.001485148,0.0001725721],"domain_scores_gemma":[0.9754362,0.01690466,0.0006747269,0.0007589075,0.005511897,0.0007135106],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001483832,0.000260166,0.002421733,0.004069226,0.00003631054,0.0001063712,0.001718133,0.0008806461,0.002000982,0.1166372,0.01970873,0.8520121],"study_design_scores_gemma":[0.00006268818,0.0005440139,0.005073803,0.006446729,0.00008273104,0.001443297,0.004610653,0.010231,0.003178962,0.1008524,0.8673018,0.0001718857],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.009099871,0.8596975,0.03654914,0.05116584,0.001408803,0.0001279344,0.00009990179,0.000631294,0.04121969],"genre_scores_gemma":[0.1150635,0.7593763,0.08904625,0.01435143,0.006195747,0.0004713548,0.0002997554,0.0003309435,0.01486471],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.00918144,"threshold_uncertainty_score":0.04855669,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2251442452","doi":"10.3115/v1/p14-1115","title":"Abstractive Summarization of Spoken and Written Conversations Based on Phrasal Queries","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Automatic summarization; Computer science; Natural language processing; Multi-document summarization; Artificial intelligence; Information retrieval","authors":[{"name":"Yashar Mehdad","is_ca":true},{"name":"Giuseppe Carenini","is_ca":true},{"name":"Raymond T. Ng","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.005305981004063103,"gpt":0.2309992697379352,"spread":0.2256932887338721,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001228127,0.002045006,0.001449911,0.002689815,0.0008158851,0.00191447,0.001371024,0.000820724,0.002716511],"category_scores_gemma":[0.005340714,0.0005118872,0.001130927,0.001743123,0.0004414082,0.002818134,0.001450888,0.001270958,0.002788189],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005662598,"about_ca_system_score_gemma":0.000986643,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003497835,"about_ca_topic_score_gemma":0.004452844,"domain_scores_codex":[0.9982669,0.0005119827,0.0001778631,0.0004317707,0.0005026745,0.0001086879],"domain_scores_gemma":[0.9960967,0.001390168,0.0004944618,0.0005178853,0.001361863,0.0001389381],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001041188,0.0002678114,0.002093774,0.00132153,0.0002878458,0.000550044,0.002988538,0.0223169,0.1832419,0.007204034,0.02664908,0.7520373],"study_design_scores_gemma":[0.0001313602,0.0008437045,0.007159353,0.0001366356,0.0007221958,0.0006336973,0.00206622,0.7779591,0.1259326,0.02580385,0.05841325,0.0001980361],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04175871,0.001195609,0.9394748,0.0004993344,0.0001938188,0.0004030476,0.002336235,0.01196198,0.002176482],"genre_scores_gemma":[0.2611777,0.001104617,0.7110545,0.000334409,0.0006012126,0.0005629639,0.01617108,0.001380401,0.007613209],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003497835,"threshold_uncertainty_score":0.009087622,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2806157614","doi":"10.1075/term.00012.amj","title":"Distributed specificity for automatic terminology extraction","year":2018,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University; Carleton University; University of Ottawa","funders":"","keywords":"Computer science; Terminology; Artificial intelligence; Classifier (UML); Filter (signal processing); Natural language processing; Representation (politics); Domain (mathematical analysis); Pattern recognition (psychology); Computer vision; Linguistics; Mathematics","authors":[{"name":"Ehsan Amjadian","is_ca":true},{"name":"Diana Inkpen","is_ca":true},{"name":"T. Sima Paribakht","is_ca":true},{"name":"Farahnaz Faez","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0178733567441031,"gpt":0.3525913017325484,"spread":0.3347179449884453,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00237627,0.0009913505,0.001078429,0.005446508,0.000901314,0.002865026,0.001210839,0.001074287,0.003311928],"category_scores_gemma":[0.01181578,0.0004536994,0.00108803,0.004297035,0.001036539,0.004452992,0.003352716,0.001750808,0.003026148],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000870625,"about_ca_system_score_gemma":0.001325697,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000832958,"about_ca_topic_score_gemma":0.001165896,"domain_scores_codex":[0.9966226,0.001326659,0.0002872752,0.0008520795,0.0006975078,0.0002138606],"domain_scores_gemma":[0.9931479,0.002975618,0.000680967,0.001621379,0.001395846,0.0001782438],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002395529,0.0001074466,0.003834964,0.0004003486,0.0001291936,0.0002246335,0.0006224069,0.009597206,0.06565396,0.03488251,0.007645292,0.8766625],"study_design_scores_gemma":[0.00009367132,0.0002695493,0.007945208,0.0002072596,0.0002182325,0.001360352,0.0009424115,0.6510552,0.08586602,0.2060213,0.04589326,0.0001275092],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02484891,0.000444797,0.9689506,0.0002543571,0.00006028469,0.0001052607,0.0003088927,0.002403185,0.002623689],"genre_scores_gemma":[0.3636183,0.0003375965,0.6291032,0.0002082988,0.0001626884,0.0002486179,0.002292338,0.0005003171,0.003528541],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005446508,"threshold_uncertainty_score":0.0125671,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2140162241","doi":"10.1007/s11192-011-0589-1","title":"Author disambiguation using multi-aspect similarity indicators","year":2011,"lang":"en","type":"article","venue":"Scientometrics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Genome Canada","keywords":"Similarity (geometry); Recall; Precision and recall; Computer science; Task (project management); Measure (data warehouse); Information retrieval; Key (lock); Data mining; Artificial intelligence; Natural language processing; Psychology; Cognitive psychology","authors":[{"name":"Thomas Gurney","is_ca":false},{"name":"Edwin Horlings","is_ca":false},{"name":"Peter van den Besselaar","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2003679371428773,"gpt":0.3850923397812263,"spread":0.184724402638349,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.005651432,0.001126165,0.001781081,0.03812621,0.001358693,0.004927465,0.00117234,0.001235099,0.001667442],"category_scores_gemma":[0.03794294,0.0004295238,0.001262913,0.03263921,0.0005896902,0.005424206,0.002900815,0.0008909502,0.001942701],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008656283,"about_ca_system_score_gemma":0.001597671,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001413587,"about_ca_topic_score_gemma":0.002407833,"domain_scores_codex":[0.992231,0.00157676,0.001420012,0.001421369,0.003004369,0.0003465647],"domain_scores_gemma":[0.9783392,0.009682412,0.00400272,0.002717776,0.004802476,0.0004554081],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000495093,0.0002854789,0.0906014,0.0007629356,0.0004001658,0.000327042,0.00116408,0.01482471,0.0146224,0.009140316,0.006980726,0.8603957],"study_design_scores_gemma":[0.0002004005,0.000609591,0.1461958,0.0003395592,0.0005456387,0.002460873,0.002356302,0.6328462,0.09661568,0.06769492,0.04969924,0.000435803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2593812,0.00233163,0.7141699,0.0005065639,0.0002674919,0.0005225941,0.004313432,0.009008055,0.009499171],"genre_scores_gemma":[0.5136863,0.0007228502,0.477934,0.00004493327,0.0001825694,0.0002801011,0.005025863,0.0004256105,0.001697849],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9618738,"threshold_uncertainty_score":0.02988797,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2066433987","doi":"10.1371/journal.pone.0071914","title":"Connected Text Reading and Differences in Text Reading Fluency in Adult Readers","year":2013,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"FP7 People: Marie-Curie Actions; National Science Foundation","keywords":"Reading (process); Fluency; Computer science; Cognitive psychology; Cognition; Reading comprehension; Psychology; Linguistics; Artificial intelligence; Mathematics education","authors":[{"name":"Sebastian Wallot","is_ca":false},{"name":"Geoff Hollis","is_ca":true},{"name":"Marieke van Rooij","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03104002281181654,"gpt":0.2376431754671723,"spread":0.2066031526553558,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007329502,0.0002411111,0.0002642628,0.001384546,0.0001724451,0.001046903,0.000194711,0.0006539963,0.004098714],"category_scores_gemma":[0.009643594,0.0001464614,0.0001701423,0.0003874338,0.0005397589,0.001124832,0.0005148234,0.0003336638,0.0007546548],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001283084,"about_ca_system_score_gemma":0.00006012833,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006621299,"about_ca_topic_score_gemma":0.0008014808,"domain_scores_codex":[0.9995996,0.00007349948,0.00006945159,0.0001283247,0.00008331632,0.00004581469],"domain_scores_gemma":[0.991397,0.004785871,0.002220077,0.000408766,0.0005834748,0.0006047193],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002272465,0.0006284341,0.924098,0.0001084989,0.0001340699,0.0007916124,0.01395043,0.0002249498,0.02654446,0.0002294214,0.0002353869,0.03078213],"study_design_scores_gemma":[0.000009419763,0.0004964896,0.9963929,0.000005599411,0.00001469904,0.0003970029,0.001065086,0.0002452469,0.001139648,0.0001512164,0.00007444822,0.0000081149],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9993547,0.00004329647,0.00009836179,0.000007668811,0.00000178677,0.000003972813,0.00003102121,0.000007038015,0.0004522416],"genre_scores_gemma":[0.9994081,0.00002049028,0.0001022512,0.000006917525,0.000004336322,0.000006070737,0.00005374184,0.000004354212,0.0003937171],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004098714,"threshold_uncertainty_score":0.01371157,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3014604514","doi":"10.2196/17642","title":"Using Natural Language Processing Techniques to Provide Personalized Educational Materials for Chronic Disease Patients in China: Development and Assessment of a Knowledge-Based Health Recommender System","year":2020,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Ontology; Computer science; Recommender system; Information retrieval; Artificial intelligence","authors":[{"name":"Zheyu Wang","is_ca":false},{"name":"Haoce Huang","is_ca":false},{"name":"Liping Cui","is_ca":false},{"name":"Juan Chen","is_ca":false},{"name":"Jiye An","is_ca":false},{"name":"Huilong Duan","is_ca":false},{"name":"Huiqing Ge","is_ca":false},{"name":"Ning Deng","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02777049579753333,"gpt":0.3899926387165861,"spread":0.3622221429190527,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002209138,0.0004822417,0.0005124989,0.0009825816,0.0005356327,0.0006715463,0.0008181354,0.0006373953,0.0009027541],"category_scores_gemma":[0.004017827,0.0001997048,0.0005860651,0.0006209242,0.0002152359,0.001039405,0.000478765,0.0003384763,0.0002915674],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001140046,"about_ca_system_score_gemma":0.002142891,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0324318,"about_ca_topic_score_gemma":0.02859724,"domain_scores_codex":[0.9991834,0.0002156394,0.0001268866,0.0002088644,0.000206325,0.00005894084],"domain_scores_gemma":[0.9980395,0.0008736587,0.0001282936,0.000158864,0.0007161378,0.00008360877],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007198158,0.001975338,0.08721629,0.0009941171,0.0002947433,0.001198384,0.001753925,0.02886768,0.049755,0.001129331,0.005388393,0.8207071],"study_design_scores_gemma":[0.0003425601,0.002184836,0.09433893,0.0001296727,0.0008522777,0.0008468785,0.001732436,0.8454027,0.04322018,0.0008662653,0.009923846,0.0001594513],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9242926,0.0005746335,0.06853702,0.0007755049,0.00005007456,0.0009334297,0.000648318,0.001812951,0.002375455],"genre_scores_gemma":[0.7813612,0.0006596756,0.2126028,0.0002620912,0.00002515503,0.0004161712,0.001767885,0.00003618595,0.002868963],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0324318,"threshold_uncertainty_score":0.06448603,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2398387993","doi":"","title":"CLASSY 2011 at TAC: Guided and Multi-lingual Summaries and Evaluation Metrics.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Computer science","authors":[{"name":"John M. Conroy","is_ca":false},{"name":"Judith D. Schlesinger","is_ca":false},{"name":"Jeff Kubina","is_ca":false},{"name":"Peter A. Rankel","is_ca":false},{"name":"Dianne P. O’Leary","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04780621551233317,"gpt":0.324504377454933,"spread":0.2766981619425999,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01770492,0.002707414,0.002058568,0.01133262,0.002273267,0.004768879,0.002762008,0.003289738,0.01193511],"category_scores_gemma":[0.08678232,0.000625378,0.00120869,0.006980216,0.0007554045,0.005107658,0.00371378,0.002476466,0.01015366],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001981159,"about_ca_system_score_gemma":0.003660043,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01270834,"about_ca_topic_score_gemma":0.02035068,"domain_scores_codex":[0.9653577,0.01565968,0.004294547,0.002717292,0.01080157,0.001169135],"domain_scores_gemma":[0.9086795,0.02606882,0.004217949,0.01788914,0.03859453,0.004550027],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001702747,0.001084475,0.00496067,0.00248579,0.0006999058,0.0002335356,0.001191923,0.007203854,0.01699727,0.002584208,0.6663132,0.2945424],"study_design_scores_gemma":[0.003404602,0.004986094,0.0590221,0.0009145178,0.001028682,0.00129146,0.003097641,0.2283819,0.0966106,0.01253283,0.587652,0.00107761],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.2329898,0.009640289,0.2136193,0.004503454,0.006311832,0.008292551,0.2733511,0.195323,0.0559687],"genre_scores_gemma":[0.2425616,0.0008229737,0.2649462,0.0005975516,0.000822546,0.005227207,0.4469612,0.01259327,0.02546761],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01770492,"threshold_uncertainty_score":0.09363365,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2091597866","doi":"10.1075/ssol.1.1.06dix","title":"The scientific study of literature","year":2011,"lang":"en","type":"article","venue":"Scientific Study of Literature","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Reading (process); Context (archaeology); Cognition; Computer science; Cognitive science; Scientific literature; Domain (mathematical analysis); Event (particle physics); Psychology; Epistemology; Data science; Cognitive psychology; Neuroscience; Linguistics; History","authors":[{"name":"Peter Dixon","is_ca":true},{"name":"Marisa Bortolussi","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02404426383237939,"gpt":0.2786062161423486,"spread":0.2545619523099692,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02227672,0.0006164649,0.00118189,0.0185078,0.003706823,0.01284808,0.001750638,0.002886347,0.005049037],"category_scores_gemma":[0.06882325,0.0003735019,0.000890119,0.01056125,0.01756027,0.01283121,0.003053026,0.00460439,0.00177564],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002947276,"about_ca_system_score_gemma":0.005752117,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008603008,"about_ca_topic_score_gemma":0.001176781,"domain_scores_codex":[0.9713039,0.01621834,0.002604535,0.001683436,0.007838313,0.0003514297],"domain_scores_gemma":[0.8285304,0.1403342,0.007579305,0.007030412,0.01431157,0.002214173],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00004347844,0.00003355936,0.001217542,0.005732081,0.00009691479,0.0003484151,0.01135699,0.0003246686,0.0007116737,0.5521671,0.1317585,0.296209],"study_design_scores_gemma":[0.000009752346,0.00003778428,0.001394523,0.004731976,0.00003000773,0.0005556417,0.003776447,0.0003331947,0.0003089852,0.2482782,0.7405103,0.0000330871],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.006198843,0.60892,0.05547628,0.1640941,0.08737712,0.0002581558,0.0004543987,0.0002994697,0.07692164],"genre_scores_gemma":[0.1488516,0.5004076,0.1038025,0.02520113,0.1940865,0.0007708053,0.0007634595,0.0003865738,0.0257299],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02227672,"threshold_uncertainty_score":0.1178119,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2163490678","doi":"","title":"Use of Keyphrase Extraction Software for Creation of an AEC/FM Thesaurus","year":2000,"lang":"en","type":"article","venue":"NPARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false},"ca_institutions":"","funders":"","keywords":"Thesaurus; Extractor; Computer science; Software; The Internet; Process (computing); Domain (mathematical analysis); Information retrieval; World Wide Web; Software engineering; Natural language processing; Engineering","authors":[{"name":"Branka Kosovac","is_ca":false},{"name":"Dana J. Vanier","is_ca":false},{"name":"Thomas Froese","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02853358991606894,"gpt":0.3131675310217468,"spread":0.2846339411056779,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003619203,0.001673528,0.001294878,0.01162616,0.001719408,0.002654992,0.0009423865,0.0009737891,0.01154616],"category_scores_gemma":[0.01391208,0.001121024,0.001536525,0.007549906,0.0008100573,0.002939845,0.001683973,0.001745355,0.007860834],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009826481,"about_ca_system_score_gemma":0.002101892,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00310612,"about_ca_topic_score_gemma":0.003259528,"domain_scores_codex":[0.998265,0.0003273671,0.0004480761,0.0004450889,0.0004536922,0.00006097325],"domain_scores_gemma":[0.9935941,0.003272327,0.000437583,0.0008081479,0.00175833,0.0001295159],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002446287,0.0001099319,0.001538824,0.001948245,0.0001656801,0.0009047514,0.003210379,0.001711564,0.06611757,0.01072498,0.01187109,0.9014524],"study_design_scores_gemma":[0.0004110356,0.0006805463,0.01159548,0.0008131212,0.0006748285,0.006310679,0.002773682,0.1236002,0.3001527,0.02951874,0.5229172,0.0005517448],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008685693,0.000220022,0.9615673,0.0001178506,0.0000890117,0.001044079,0.002056347,0.0221804,0.004039361],"genre_scores_gemma":[0.0129864,0.0001380504,0.980975,0.00003140825,0.00001766748,0.0005722399,0.002145686,0.0014901,0.001643442],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01162616,"threshold_uncertainty_score":0.03862578,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2086368750","doi":"10.1145/371920.372162","title":"When experts agree","year":2001,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto","funders":"","keywords":"Watson; Citation; Library science; Research center; George (robot); Center (category theory); Computer science; World Wide Web; Operations research; Engineering; Political science; Artificial intelligence","authors":[{"name":"Krishna Bharat","is_ca":false},{"name":"George A. Mihaila","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0188195129517892,"gpt":0.2764848424940115,"spread":0.2576653295422223,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.03607197,0.001345716,0.001736841,0.00511602,0.007934975,0.02195543,0.004636337,0.02568681,0.4102354],"category_scores_gemma":[0.2671683,0.001192534,0.002029302,0.003369955,0.003203348,0.01946336,0.01641389,0.01520382,0.3477953],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004726788,"about_ca_system_score_gemma":0.01715548,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002312144,"about_ca_topic_score_gemma":0.004203663,"domain_scores_codex":[0.9359561,0.01358089,0.006567724,0.004790775,0.03010546,0.008999176],"domain_scores_gemma":[0.7451984,0.03410846,0.008736487,0.0221131,0.1583264,0.03151704],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002556111,0.00001235102,0.0001337129,0.00008035214,0.000006157115,0.00006848673,0.0002701085,0.00001768661,0.0001044995,0.001195946,0.9836777,0.01440742],"study_design_scores_gemma":[0.00002515303,0.00001401528,0.0002641361,0.0002570184,0.000008688387,0.00005374502,0.001981174,0.00007385023,0.000191658,0.003989807,0.9930958,0.00004487902],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.002605378,0.001635931,0.007721998,0.4743256,0.1761157,0.002091706,0.004296521,0.003861675,0.3273455],"genre_scores_gemma":[0.02824222,0.002163771,0.01123708,0.2430878,0.05459128,0.004868158,0.0046252,0.004096953,0.6470875],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4102354,"threshold_uncertainty_score":0.8412276,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1978683092","doi":"10.1177/016224390202700301","title":"From Thing to Sign and “Natural Object”: Toward a Genetic Phenomenology of Graph Interpretation","year":2002,"lang":"en","type":"article","venue":"Science Technology & Human Values","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Trois-Rivières; Lakehead University; University of Victoria","funders":"","keywords":"Interpretation (philosophy); Epistemology; Natural science; Phenomenology (philosophy); Reading (process); Natural (archaeology); Object (grammar); Sign (mathematics); Computer science; Semiotics; Cognitive science; Sociology; Linguistics; Psychology; Artificial intelligence; Mathematics; Philosophy; History","authors":[{"name":"Wolff‐Michael Roth","is_ca":true},{"name":"G. Michael Bowen","is_ca":true},{"name":"Domenico Masciotra","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01471340560740698,"gpt":0.2846390009863392,"spread":0.2699255953789322,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.004943949,0.0003687608,0.0002991229,0.002047853,0.002877535,0.006969021,0.001226023,0.00209269,0.002166449],"category_scores_gemma":[0.01339968,0.0004609296,0.0004728779,0.001489212,0.03576066,0.01488666,0.003460974,0.003299693,0.0002771814],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002768555,"about_ca_system_score_gemma":0.00153267,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003232219,"about_ca_topic_score_gemma":0.002277271,"domain_scores_codex":[0.9964563,0.002697813,0.00005896352,0.0003566817,0.0002890571,0.0001412505],"domain_scores_gemma":[0.9929367,0.005299094,0.0005035005,0.0006227784,0.0003474434,0.0002905451],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00005419944,0.00004665983,0.003745286,0.00007643113,0.000008906976,0.0004499087,0.3564538,0.0008586989,0.002830691,0.6266785,0.0004977694,0.008299088],"study_design_scores_gemma":[0.00002801196,0.00006574934,0.002371939,0.00008810208,0.00001641194,0.000607689,0.1626403,0.00617307,0.001470295,0.8051632,0.02132424,0.00005102488],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5593387,0.0007694038,0.3242385,0.01572191,0.0001478806,0.0002640172,0.0001710809,0.0002270843,0.09912146],"genre_scores_gemma":[0.9687222,0.0001817736,0.02861913,0.0003459829,0.00001362047,0.00006408285,0.00004778547,0.00007910611,0.001926329],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9971225,"threshold_uncertainty_score":0.02614641,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}