{"meta":{"query_hash":"b13bc4f485d2","filters":{"topic":"Advanced Text Analysis Techniques"},"cohort_total":552,"direct_labels_cover":2,"predictions_cover":552,"exported":552,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/b13bc4f485d2","api":"https://metacan.xera.ac/api/v1/cohort?topic=Advanced+Text+Analysis+Techniques"},"results":[{"id":"W103177335","doi":"10.7202/1032997ar","title":"Indexation manuelle et indexation assistée par ordinateur : comparaison de la performance de deux index d’une monographie","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Indexation","score_opus":0.03547252047443438,"score_gpt":0.35930764876178134,"score_spread":0.32383512828734695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W103177335","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.447131,0.010502484,0.39859477,0.0017629459,0.0012654599,0.0016797755,0.0108269695,0.07795947,0.050277114],"genre_scores_gemma":[0.39056253,0.002095083,0.5517852,0.00023718279,0.00022729542,0.000638514,0.011912035,0.003439016,0.039103154],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99446183,0.0013436177,0.0006421296,0.00096661487,0.0023480623,0.00023762707],"domain_scores_gemma":[0.975593,0.012406058,0.001216981,0.0036327543,0.006497706,0.00065363327],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0045995996,0.0011450573,0.001451876,0.0072391825,0.0012508323,0.0055096974,0.0012408067,0.001059316,0.013705015],"category_scores_gemma":[0.03374565,0.00053944875,0.00087752554,0.0067720986,0.0006458477,0.004298902,0.0019623656,0.000859699,0.0077859974],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017101277,0.00028781858,0.008511181,0.0021733677,0.00023243355,0.00021357513,0.0025825226,0.0049105557,0.046508886,0.0042112614,0.018298006,0.9103602],"study_design_scores_gemma":[0.0006490038,0.002836368,0.08097,0.0008394561,0.0007419204,0.0015179379,0.005871215,0.21203706,0.2658594,0.010698892,0.4172744,0.0007043506],"about_ca_topic_score_codex":0.014773257,"about_ca_topic_score_gemma":0.011589871,"teacher_disagreement_score":0.9944903,"about_ca_system_score_codex":0.0013282582,"about_ca_system_score_gemma":0.0026218514,"threshold_uncertainty_score":0.045847833},"labels":[],"label_agreement":null},{"id":"W112330709","doi":"10.1007/978-3-642-14616-9_38","title":"A Survey of Text Extraction Tools for Intelligent Healthcare Decision Support Systems","year":2010,"lang":"en","type":"book-chapter","venue":"Smart innovation, systems and technologies","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Decision support system; Field (mathematics); Data science; Newspaper; Domain (mathematical analysis); Process (computing); Intelligent decision support system; Health care; Domain knowledge; Information extraction; Knowledge management; Artificial intelligence","score_opus":0.07491786707459162,"score_gpt":0.33351585440880405,"score_spread":0.2585979873342124,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W112330709","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027561871,0.22535358,0.66281486,0.0041349153,0.0009949823,0.001185137,0.029455505,0.026843678,0.02165549],"genre_scores_gemma":[0.044479575,0.13393421,0.7451076,0.0021012311,0.0008436167,0.0008904002,0.05021157,0.0026050531,0.01982683],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982582,0.00030947826,0.00039796953,0.0003286076,0.0006322894,0.00007358559],"domain_scores_gemma":[0.99109685,0.0065004462,0.00045592553,0.00040461274,0.0014127039,0.00012951411],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022966943,0.0014842284,0.0017583016,0.011010232,0.000828965,0.0034022569,0.0015869756,0.00091685826,0.006544577],"category_scores_gemma":[0.0075130975,0.00066354935,0.0013561635,0.013014285,0.00036520595,0.0046123583,0.00096996635,0.00092669023,0.0054335156],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009469497,0.00008068558,0.0011948935,0.003895381,0.00009936864,0.00016741113,0.00025416425,0.0006896821,0.006666672,0.0018721455,0.031977795,0.95300716],"study_design_scores_gemma":[0.00009562644,0.00035273956,0.013669688,0.005630745,0.0008535764,0.003197109,0.0010364439,0.033691566,0.06772946,0.019253798,0.8542562,0.00023310972],"about_ca_topic_score_codex":0.0016077437,"about_ca_topic_score_gemma":0.0025120643,"teacher_disagreement_score":0.011010232,"about_ca_system_score_codex":0.000588664,"about_ca_system_score_gemma":0.0014349205,"threshold_uncertainty_score":0.021893859},"labels":[],"label_agreement":null},{"id":"W115958420","doi":"","title":"Comparing Out-of-Sample Predictive Ability of PLS, Covariance, and Regression Models","year":2014,"lang":"en","type":"article","venue":"QUT ePrints (Queensland University of Technology)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Partial least squares regression; Covariance; Structural equation modeling; Regression; Regression analysis; Computer science; Sample (material); Predictive modelling; Analysis of covariance; Range (aeronautics); Econometrics; Statistics; Machine learning; Artificial intelligence; Mathematics; Engineering","score_opus":0.01629929383035682,"score_gpt":0.23156226924726164,"score_spread":0.2152629754169048,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W115958420","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5247743,0.002614306,0.45517,0.0026366378,0.0003680516,0.0005618283,0.00123377,0.0018120861,0.010829039],"genre_scores_gemma":[0.92623854,0.0008786396,0.069274426,0.00023248795,0.00012078605,0.00033929636,0.0015531972,0.00037082445,0.0009918398],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9613788,0.032692183,0.0009326923,0.0020019508,0.002528017,0.00046638196],"domain_scores_gemma":[0.4514617,0.5091751,0.008006504,0.019577473,0.010630479,0.0011486984],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.100956,0.002121313,0.0018037424,0.0037629067,0.0011871041,0.003534344,0.0022583038,0.0019032712,0.003578],"category_scores_gemma":[0.33731422,0.0007320079,0.0025866122,0.0046051787,0.0024295603,0.008773762,0.0031850366,0.0036363653,0.00089225546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025023215,0.0011595284,0.12858932,0.001289128,0.0037440297,0.00032049406,0.004038935,0.47888568,0.00095382525,0.09749602,0.009153898,0.27186692],"study_design_scores_gemma":[0.00015023227,0.0006118797,0.019938797,0.00025091408,0.0004378096,0.00010506503,0.00088642206,0.911537,0.0010463282,0.06241224,0.0025150788,0.0001082614],"about_ca_topic_score_codex":0.0070613544,"about_ca_topic_score_gemma":0.0066112233,"teacher_disagreement_score":0.100956,"about_ca_system_score_codex":0.0014517059,"about_ca_system_score_gemma":0.0026535783,"threshold_uncertainty_score":0.5339128},"labels":[],"label_agreement":null},{"id":"W135208616","doi":"","title":"Bayesian Structural Equation Models for Cumulative Theory Building in Information Systems.","year":2012,"lang":"en","type":"article","venue":"Journal of the Association for Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Computer science; Structural equation modeling; Latent variable; Bayesian probability; Proxy (statistics); Reuse; Item response theory; Context (archaeology); Statistical model; Econometrics; Information theory; Machine learning; Data mining; Artificial intelligence; Mathematics; Statistics; Engineering","score_opus":0.022186923798461936,"score_gpt":0.29285704316334377,"score_spread":0.27067011936488183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W135208616","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022867701,0.0022876689,0.9881039,0.0022907732,0.000117510106,0.00022925863,0.0006019079,0.00025139423,0.0038307207],"genre_scores_gemma":[0.18993543,0.0068485825,0.78933954,0.0013217983,0.00065663614,0.00339558,0.0023106034,0.00023397748,0.0059579085],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9800087,0.014381762,0.00078755134,0.0018610131,0.0025595154,0.00040145146],"domain_scores_gemma":[0.92742324,0.06177634,0.004796449,0.0028803395,0.0025771214,0.0005466202],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02194852,0.0024484443,0.0029959239,0.00707246,0.0018925762,0.0053668735,0.005505926,0.0043025957,0.0120454915],"category_scores_gemma":[0.09523896,0.0017762556,0.0031444617,0.010154362,0.00386416,0.009710856,0.004337556,0.0073174005,0.0026760167],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027294283,0.00008931785,0.0017951631,0.0002888417,0.00027245225,0.000084686064,0.00059388933,0.05941005,0.00009477345,0.8982784,0.0038665135,0.035198625],"study_design_scores_gemma":[0.000022507558,0.000031124153,0.0005172036,0.00016059744,0.00008121215,0.000054354678,0.00007608452,0.15024085,0.00004816049,0.84101534,0.0077134175,0.000039264098],"about_ca_topic_score_codex":0.012243478,"about_ca_topic_score_gemma":0.021505423,"teacher_disagreement_score":0.02194852,"about_ca_system_score_codex":0.0064341747,"about_ca_system_score_gemma":0.0038911707,"threshold_uncertainty_score":0.11607623},"labels":[],"label_agreement":null},{"id":"W1486865875","doi":"10.48550/arxiv.cs/0308033","title":"Coherent Keyphrase Extraction via Web Mining","year":2003,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":160,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Extraction (chemistry); Web mining; World Wide Web; Web page","score_opus":0.04399676745781409,"score_gpt":0.3191097352841954,"score_spread":0.2751129678263813,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1486865875","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031398665,0.0013201032,0.95329773,0.00033659098,0.00008056884,0.0005689046,0.0020993757,0.0082772765,0.0026207273],"genre_scores_gemma":[0.123245955,0.001027868,0.8659865,0.000120085235,0.00012257427,0.0003417538,0.0061328476,0.00059825415,0.002424122],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970228,0.0005550193,0.00041713158,0.00080348493,0.0010102886,0.00019123245],"domain_scores_gemma":[0.99107766,0.0037546903,0.0013008497,0.0015419904,0.0021569463,0.00016777076],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023480777,0.0016119557,0.0013395292,0.013080889,0.0012068362,0.0029833254,0.0014728949,0.001349852,0.0031047359],"category_scores_gemma":[0.015647123,0.00079222844,0.0015489829,0.0106990915,0.0008047823,0.004820709,0.0021424142,0.0013053331,0.0047925464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043654398,0.00022886001,0.004738349,0.0011683514,0.00017390141,0.0008314447,0.0005954363,0.0073697683,0.066037565,0.0091414545,0.010557164,0.8987212],"study_design_scores_gemma":[0.00027953068,0.00048484275,0.015879154,0.00042159276,0.0005277327,0.003632981,0.0017437958,0.47298744,0.25140223,0.12988843,0.12244485,0.0003074726],"about_ca_topic_score_codex":0.0015047881,"about_ca_topic_score_gemma":0.0019092362,"teacher_disagreement_score":0.013080889,"about_ca_system_score_codex":0.00068275305,"about_ca_system_score_gemma":0.001535089,"threshold_uncertainty_score":0.012417972},"labels":[],"label_agreement":null},{"id":"W1494478641","doi":"10.1007/978-0-387-69810-6","title":"The Statistical Analysis of Recurrent Events","year":2007,"lang":"en","type":"book","venue":"Statistics for biology and health","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":738,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science","score_opus":0.04241724093842392,"score_gpt":0.43682276115090307,"score_spread":0.39440552021247915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1494478641","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005392426,0.0052614454,0.99002445,0.00044606507,0.0003603286,0.000023736642,0.00035904226,0.0011428117,0.001842854],"genre_scores_gemma":[0.04111317,0.012169791,0.9222741,0.00090860034,0.0028859803,0.00064863375,0.00345956,0.0019373384,0.014602747],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9956585,0.002000228,0.0003433472,0.0006231916,0.001286121,0.000088512534],"domain_scores_gemma":[0.9694932,0.025755696,0.00075675076,0.0024931051,0.0013042826,0.00019693073],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006793679,0.0014764569,0.0021707625,0.0031476126,0.00037167498,0.0022010072,0.002010452,0.0011721288,0.0064228177],"category_scores_gemma":[0.03536831,0.00096597214,0.0015988933,0.0037669274,0.002327925,0.0029737004,0.0011797223,0.003842383,0.005357296],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007544936,0.000047870388,0.0012842718,0.0008406391,0.00033338682,0.00022259155,0.00030346762,0.0147597175,0.0024430782,0.22891626,0.099686965,0.6510864],"study_design_scores_gemma":[0.00002509755,0.00005879429,0.0024392947,0.00016758681,0.000114428774,0.0006036725,0.000053954886,0.11531941,0.0018967955,0.8247179,0.054529823,0.000073307296],"about_ca_topic_score_codex":0.0009019421,"about_ca_topic_score_gemma":0.0011791033,"teacher_disagreement_score":0.006793679,"about_ca_system_score_codex":0.00070303254,"about_ca_system_score_gemma":0.001114642,"threshold_uncertainty_score":0.035928845},"labels":[],"label_agreement":null},{"id":"W1495605239","doi":"10.18438/b8v03n","title":"The Usefulness of Related Functions in Web of Science and Scopus","year":2011,"lang":"en","type":"article","venue":"Evidence Based Library and Information Practice","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Scopus; Web of science; Information retrieval; Relevance (law); Computer science; Wilcoxon signed-rank test; Mathematics; Statistics; Medicine; MEDLINE; Mann–Whitney U test; Meta-analysis; Internal medicine","score_opus":0.02176733728420697,"score_gpt":0.25292333423715113,"score_spread":0.23115599695294417,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1495605239","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9373495,0.015549959,0.011681917,0.0011121054,0.00019057805,0.00059050944,0.0026864538,0.0010833949,0.029755466],"genre_scores_gemma":[0.98559386,0.002042881,0.010206947,0.00008051279,0.00013856839,0.00013297237,0.0011904998,0.00007046656,0.00054343604],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9739551,0.008290915,0.0035219654,0.001432094,0.012231456,0.00056854],"domain_scores_gemma":[0.7778708,0.1754264,0.020376762,0.0063827224,0.01766939,0.0022739975],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.021091392,0.0007653459,0.0014621825,0.025800321,0.00079106103,0.0046200794,0.00080614816,0.0009295147,0.0024957482],"category_scores_gemma":[0.19600642,0.0002692131,0.0014705256,0.01523241,0.00075970247,0.005578116,0.0018762596,0.00049599103,0.00089688465],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038209883,0.0006581227,0.33682063,0.007234281,0.0016991339,0.000549216,0.0017018868,0.0045453124,0.004401111,0.0022854055,0.0063002165,0.6299836],"study_design_scores_gemma":[0.0004947867,0.0041347244,0.931283,0.0022976683,0.0020450996,0.0028641534,0.0034560773,0.02440693,0.012028932,0.0047882544,0.011862345,0.0003379888],"about_ca_topic_score_codex":0.0024137625,"about_ca_topic_score_gemma":0.0016312452,"teacher_disagreement_score":0.9789086,"about_ca_system_score_codex":0.0014030066,"about_ca_system_score_gemma":0.0018038077,"threshold_uncertainty_score":0.11154324},"labels":[],"label_agreement":null},{"id":"W1499244104","doi":"10.1007/978-3-642-24469-8_27","title":"Making Sense in the Margins: A Field Study of Annotation","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nipissing University; Dalhousie University","funders":"","keywords":"Annotation; Computer science; Hypertext; Field (mathematics); Style (visual arts); Information retrieval; World Wide Web; Artificial intelligence","score_opus":0.040288491415573,"score_gpt":0.3078085936347314,"score_spread":0.2675201022191584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1499244104","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8345612,0.0011765668,0.03251556,0.0034302538,0.00016430246,0.00040821216,0.00030857802,0.00019263187,0.12724271],"genre_scores_gemma":[0.98365027,0.0002684175,0.006212628,0.00025709585,0.000050536953,0.00014277207,0.00016300558,0.00017488847,0.009080327],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9879253,0.008716979,0.0004250259,0.0012539413,0.001297502,0.00038125718],"domain_scores_gemma":[0.8874903,0.0952747,0.003543862,0.0060750735,0.006402519,0.0012134745],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015019883,0.0004400952,0.00041220535,0.0031934043,0.007081574,0.005639825,0.0019046492,0.0017754339,0.0078829955],"category_scores_gemma":[0.07626405,0.00052735774,0.00024298183,0.0035827172,0.010291638,0.013719694,0.005095149,0.003039216,0.0010790644],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002350333,0.00015862251,0.009791281,0.00031563023,0.000007254207,0.00027789647,0.879454,0.00010625906,0.004494444,0.047200058,0.002768111,0.05519137],"study_design_scores_gemma":[0.000053583855,0.0002530643,0.027328549,0.00083165895,0.00003585257,0.0009200649,0.7751671,0.0025157265,0.0076288567,0.07931697,0.10584974,0.00009884929],"about_ca_topic_score_codex":0.005483998,"about_ca_topic_score_gemma":0.005241859,"teacher_disagreement_score":0.015019883,"about_ca_system_score_codex":0.002540555,"about_ca_system_score_gemma":0.002609322,"threshold_uncertainty_score":0.07943368},"labels":[],"label_agreement":null},{"id":"W1499800862","doi":"10.1007/3-540-45637-6_4","title":"Extracting Keyphrases from Spoken Audio Documents","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Utterance; Natural language processing; Speech recognition; Artificial intelligence; Robustness (evolution); Transcription (linguistics); Word error rate; Linguistics","score_opus":0.01822006719168502,"score_gpt":0.2707262638742662,"score_spread":0.2525061966825812,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1499800862","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21449472,0.007495635,0.7286061,0.00048471856,0.0010015371,0.00085453654,0.01101413,0.021672582,0.014376085],"genre_scores_gemma":[0.34968835,0.0051382403,0.60003996,0.00013287908,0.0005587082,0.00037017427,0.017469347,0.0013253101,0.025276998],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995442,0.000033932836,0.000047013775,0.00014167797,0.00014978468,0.00008327913],"domain_scores_gemma":[0.9988242,0.00044939556,0.00012178957,0.00014183756,0.00038159793,0.000081173464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00026797754,0.0015556063,0.0010313679,0.0032894053,0.00060357014,0.0016832215,0.0007393682,0.0010193484,0.01163925],"category_scores_gemma":[0.0019239454,0.00047806112,0.0008215393,0.0027271565,0.00040099834,0.0018460749,0.00097448146,0.0008286238,0.019242298],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007339625,0.00005613338,0.000771887,0.00077563117,0.00003523712,0.00083773036,0.00024557143,0.0005795102,0.36766124,0.00092485955,0.0050455984,0.6223327],"study_design_scores_gemma":[0.00035396437,0.0016093048,0.026200203,0.0003570278,0.0007418382,0.007942563,0.003343604,0.09566601,0.7446875,0.0081321485,0.11074643,0.000219393],"about_ca_topic_score_codex":0.0020677655,"about_ca_topic_score_gemma":0.0030409822,"teacher_disagreement_score":0.01163925,"about_ca_system_score_codex":0.00037590344,"about_ca_system_score_gemma":0.0007085017,"threshold_uncertainty_score":0.03893715},"labels":[],"label_agreement":null},{"id":"W1506516332","doi":"10.1007/978-3-540-73351-5_22","title":"Combining Vector Space Model and Multi Word Term Extraction for Semantic Query Expansion","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Ranking (information retrieval); Computer science; Vector space model; Margin (machine learning); Artificial intelligence; Semantic similarity; Term (time); Word (group theory); Natural language processing; Cluster analysis; Similarity (geometry); Space (punctuation); Vector space; Information retrieval; Machine learning; Mathematics","score_opus":0.03547433304624778,"score_gpt":0.3152286062239996,"score_spread":0.2797542731777518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1506516332","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0141663775,0.002089459,0.97533196,0.0002752573,0.00015525053,0.00015568701,0.0008989827,0.005822051,0.0011048801],"genre_scores_gemma":[0.191905,0.0022545394,0.7927894,0.00031889515,0.00032030072,0.0004146193,0.0064294157,0.0008066501,0.0047611697],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979977,0.0005608291,0.00023292821,0.0004274407,0.0006356506,0.00014542454],"domain_scores_gemma":[0.998092,0.0009552284,0.000080656995,0.00025213644,0.0005762929,0.00004363785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017942156,0.0014813042,0.0024631443,0.005132548,0.00063675473,0.0017418425,0.0014083278,0.0011510716,0.0039755465],"category_scores_gemma":[0.004086909,0.00056439626,0.0018866712,0.0069050775,0.00052935863,0.0057447394,0.0015044312,0.001476123,0.0032603475],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005787929,0.00035859665,0.0011501124,0.0006112618,0.00027514333,0.00018125714,0.00019677641,0.033131152,0.036612265,0.01155797,0.017147569,0.8981991],"study_design_scores_gemma":[0.00007253924,0.00013520062,0.00069649226,0.000031971413,0.00016899013,0.00024718474,0.000104774655,0.9563053,0.014714266,0.020594772,0.006861108,0.00006745756],"about_ca_topic_score_codex":0.007398649,"about_ca_topic_score_gemma":0.0077905073,"teacher_disagreement_score":0.007398649,"about_ca_system_score_codex":0.0007374036,"about_ca_system_score_gemma":0.0014008338,"threshold_uncertainty_score":0.014711142},"labels":[],"label_agreement":null},{"id":"W1506605920","doi":"10.1007/978-1-4020-5347-4_15","title":"TOWARDS A CHANGE-BASED CHANCE DISCOVERY","year":2009,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Knowledge base; Computer science; Seekers; Data science; Knowledge extraction; Artificial intelligence; Political science","score_opus":0.035014388041817154,"score_gpt":0.27531860109267586,"score_spread":0.2403042130508587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1506605920","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040584314,0.00060558424,0.984761,0.00090543967,0.00014100278,0.00006398178,0.000228015,0.000720798,0.008515685],"genre_scores_gemma":[0.16870455,0.0012866491,0.803156,0.0005159688,0.00047423967,0.00029369647,0.0008635702,0.0005788085,0.024126511],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967186,0.00092583,0.00016213089,0.00085731415,0.0012033886,0.00013262806],"domain_scores_gemma":[0.9877965,0.009298215,0.00043305487,0.001368419,0.0008915043,0.00021223199],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042660036,0.0007843281,0.001770202,0.003200753,0.0012683698,0.004382955,0.003944575,0.001764542,0.008594634],"category_scores_gemma":[0.023343356,0.00084175554,0.0021240483,0.0034834128,0.002757149,0.0074012126,0.004050712,0.0039664465,0.003183485],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014947422,0.0000834661,0.0037353025,0.00029126916,0.00018724897,0.00022446724,0.0007911478,0.033894755,0.0024825898,0.55855304,0.01366198,0.38594526],"study_design_scores_gemma":[0.000018454444,0.000031371146,0.0010052557,0.00005992598,0.00006117859,0.0002584526,0.00012213142,0.35360333,0.001763474,0.6301386,0.012895193,0.00004267216],"about_ca_topic_score_codex":0.0028323936,"about_ca_topic_score_gemma":0.0025729816,"teacher_disagreement_score":0.008594634,"about_ca_system_score_codex":0.0013089186,"about_ca_system_score_gemma":0.0014550427,"threshold_uncertainty_score":0.02875191},"labels":[],"label_agreement":null},{"id":"W1509964004","doi":"10.7202/1032764ar","title":"L’analyse du texte littéraire assistée par ordinateur : essai d’illustration avec Regards et jeux dans l’espace, de Saint-Denys Garneau, traité avec le logiciel SATO","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université de Montréal","funders":"","keywords":"Humanities; Art","score_opus":0.0358857120120147,"score_gpt":0.3253563984386073,"score_spread":0.2894706864265926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1509964004","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15116788,0.035779778,0.3417453,0.0663375,0.005830367,0.0006644184,0.0066460064,0.00878288,0.3830459],"genre_scores_gemma":[0.3691552,0.019845363,0.21225154,0.004670254,0.0013801204,0.000473579,0.006178841,0.005556798,0.3804883],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9976215,0.0009873842,0.00014475077,0.00034644254,0.0008163625,0.00008360197],"domain_scores_gemma":[0.99382496,0.0037047467,0.00022979724,0.00044561547,0.0016361721,0.0001587331],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022874246,0.0006351023,0.0004328142,0.004388145,0.0026038873,0.0060302946,0.0005946199,0.0013117646,0.019341169],"category_scores_gemma":[0.010517244,0.0003839969,0.0005247456,0.0038001647,0.0023555602,0.005212836,0.001967929,0.0018675134,0.0063911695],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039536826,0.00010214649,0.007375534,0.0021783595,0.00007291438,0.0039486657,0.118808344,0.0016395453,0.03822927,0.17638639,0.18267973,0.4681838],"study_design_scores_gemma":[0.000009201748,0.000027717639,0.0035045731,0.0005319924,0.000019646051,0.0010026282,0.009842832,0.0016519829,0.0043349867,0.0070102857,0.9720223,0.000041813488],"about_ca_topic_score_codex":0.021278866,"about_ca_topic_score_gemma":0.028349739,"teacher_disagreement_score":0.021278866,"about_ca_system_score_codex":0.0025786674,"about_ca_system_score_gemma":0.0031087734,"threshold_uncertainty_score":0.06470269},"labels":[],"label_agreement":null},{"id":"W1510117420","doi":"","title":"Extracting semantically-coherent keyphrases from speech","year":2004,"lang":"en","type":"article","venue":"NPARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada; University of Ottawa","funders":"University of Pennsylvania","keywords":"Computer science; Extractor; Artificial intelligence; Similarity (geometry); Word (group theory); Natural language processing; Speech recognition; Key (lock); Semantic similarity; Pattern recognition (psychology); Linguistics; Image (mathematics)","score_opus":0.015152472103937215,"score_gpt":0.27316790850334316,"score_spread":0.25801543639940594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1510117420","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17829633,0.004770712,0.77531564,0.00081229623,0.0007578716,0.00053022266,0.01457366,0.016128717,0.008814593],"genre_scores_gemma":[0.36069956,0.0034307812,0.6082352,0.00020067519,0.0007093409,0.00021222224,0.01842148,0.0015606399,0.006530085],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99924845,0.00008773122,0.00010019258,0.00025037673,0.00020644334,0.00010681996],"domain_scores_gemma":[0.99773824,0.00077438966,0.00037782136,0.00034950112,0.0006510715,0.00010890229],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00052223774,0.002211824,0.0010992833,0.0050839623,0.0006207242,0.0018121075,0.00079630606,0.0011886331,0.006274964],"category_scores_gemma":[0.003612474,0.00051978545,0.0008593791,0.0039737415,0.0007923643,0.003807528,0.0014504492,0.0010410263,0.009069244],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001313273,0.00011523434,0.0017667679,0.0018091358,0.00007312602,0.0017647456,0.001086314,0.0021020765,0.49005154,0.008633785,0.013049171,0.47823486],"study_design_scores_gemma":[0.00031793152,0.0014983596,0.020583984,0.0005228121,0.00064739195,0.006538455,0.0068325126,0.12696046,0.60154337,0.06700586,0.16714711,0.000401805],"about_ca_topic_score_codex":0.0012580886,"about_ca_topic_score_gemma":0.0017932156,"teacher_disagreement_score":0.006274964,"about_ca_system_score_codex":0.00042759065,"about_ca_system_score_gemma":0.00090226566,"threshold_uncertainty_score":0.020991862},"labels":[],"label_agreement":null},{"id":"W1510360741","doi":"10.1016/s0166-4115(08)10003-6","title":"Short- vs. Long-Term Memory","year":2008,"lang":"en","type":"book-chapter","venue":"Advances in psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Psychology; Cognitive psychology; Term (time); Long-term memory; Short-term memory; Construct (python library); Cognitive science; Cognition; Working memory; Computer science; Neuroscience","score_opus":0.02490165379644471,"score_gpt":0.3572531390102625,"score_spread":0.3323514852138178,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1510360741","genre_codex":"other","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00926309,0.17605606,0.043725893,0.004899861,0.0040172036,0.000042410466,0.00061283703,0.0005350002,0.76084757],"genre_scores_gemma":[0.122819856,0.07701685,0.018267553,0.0020393878,0.003140991,0.000081036225,0.00081682164,0.00043389364,0.77538365],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9998971,0.000012208579,0.0000073247948,0.000030097488,0.000041406707,0.0000119721935],"domain_scores_gemma":[0.9996427,0.00020843522,0.000019639463,0.00004773812,0.0000572384,0.000024236657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00027248616,0.00050213136,0.00039391997,0.00062456017,0.00033452982,0.002963346,0.0012715159,0.0008872083,0.03291215],"category_scores_gemma":[0.0010338834,0.00020055963,0.00028740158,0.0009575137,0.00088785274,0.0050631636,0.000516164,0.0012600806,0.013553133],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000933403,0.000043067495,0.00033287477,0.0005906933,0.000018756948,0.00013584377,0.0005256284,0.00023574648,0.0034180824,0.41614234,0.061684128,0.51677954],"study_design_scores_gemma":[0.000015042618,0.00006737694,0.0019323524,0.00039995668,0.000044920955,0.00091663044,0.00034438827,0.0009887273,0.004624842,0.44384518,0.54679316,0.000027408872],"about_ca_topic_score_codex":0.0005267948,"about_ca_topic_score_gemma":0.00089583744,"teacher_disagreement_score":0.03291215,"about_ca_system_score_codex":0.00055161654,"about_ca_system_score_gemma":0.0003933807,"threshold_uncertainty_score":0.11010218},"labels":[],"label_agreement":null},{"id":"W1511980079","doi":"10.1023/a:1021855607270","title":"Categorisation Techniques in Computer-Assisted Reading and Analysis of Texts (CARAT) in the Humanities","year":2003,"lang":"en","type":"article","venue":"Computers and the Humanities","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Computational linguistics; Set (abstract data type); Process (computing); Reading (process); Natural language processing; Linguistics; Artificial intelligence; Programming language; Philosophy","score_opus":0.026891381403847812,"score_gpt":0.25621449946644526,"score_spread":0.22932311806259745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1511980079","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01698106,0.0026783973,0.9627343,0.0010716753,0.00036576774,0.0008209973,0.000814171,0.0044920044,0.0100417305],"genre_scores_gemma":[0.09037993,0.0012353386,0.89791083,0.00022689936,0.00016877623,0.00085953606,0.0013836324,0.0010957241,0.0067394734],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98789537,0.0076893973,0.0009185111,0.0015507869,0.0016130168,0.00033292468],"domain_scores_gemma":[0.96139306,0.028935568,0.0012774603,0.0032390386,0.004769684,0.00038523824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072265556,0.0011297184,0.001209338,0.014109269,0.0029032573,0.0071135913,0.0023454465,0.0020487595,0.012229381],"category_scores_gemma":[0.027357323,0.00072713685,0.0017816816,0.0107309045,0.0036305182,0.00785522,0.0036189763,0.0037496984,0.0043290453],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000251472,0.00017070418,0.0015044771,0.0013043765,0.00011817197,0.0002214719,0.016072625,0.0016628541,0.012168978,0.08546381,0.01205399,0.8690071],"study_design_scores_gemma":[0.0001757506,0.00038288592,0.01266422,0.0018435674,0.00039778164,0.0019189997,0.028874343,0.092935584,0.0579101,0.47130898,0.33118945,0.00039843086],"about_ca_topic_score_codex":0.0032649145,"about_ca_topic_score_gemma":0.0038568398,"teacher_disagreement_score":0.014109269,"about_ca_system_score_codex":0.0019039252,"about_ca_system_score_gemma":0.002437739,"threshold_uncertainty_score":0.040911376},"labels":[],"label_agreement":null},{"id":"W1514107135","doi":"","title":"Building Systematic Reviews Using Automatic Text Classification Techniques","year":2010,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Workflow; Exploit; Classifier (UML); Process (computing); Protocol (science); Data mining; Machine learning; Systematic review; Artificial intelligence; Data science; Information retrieval; MEDLINE; Database","score_opus":0.05575739337578075,"score_gpt":0.36121581644412903,"score_spread":0.3054584230683483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1514107135","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009244242,0.0024892704,0.9756039,0.00086532667,0.00025370912,0.0031316543,0.0021202,0.005301302,0.0009903915],"genre_scores_gemma":[0.010243242,0.00037457928,0.98632205,0.00009917104,0.000101538586,0.0012928982,0.0011994576,0.000089810936,0.000277234],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9623158,0.018461585,0.008421906,0.0045882305,0.005849863,0.0003625215],"domain_scores_gemma":[0.79657245,0.1461726,0.015837431,0.012125336,0.02821519,0.0010770343],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.035479527,0.0022975008,0.0033805012,0.020393964,0.0018370296,0.0038370846,0.0024090905,0.0021410394,0.0044134227],"category_scores_gemma":[0.13539636,0.0012898213,0.004269015,0.0097531425,0.000654029,0.004775005,0.0020024402,0.002296714,0.003189301],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021283097,0.00021735884,0.004672793,0.006228579,0.00087044423,0.0003638082,0.00088222796,0.012030124,0.012740444,0.0041706874,0.015141987,0.94246864],"study_design_scores_gemma":[0.000987707,0.0011342687,0.014893275,0.0038787532,0.0042600296,0.0016987297,0.0014176777,0.66863847,0.06856884,0.10972599,0.124252416,0.0005438171],"about_ca_topic_score_codex":0.0021885806,"about_ca_topic_score_gemma":0.006399029,"teacher_disagreement_score":0.96452045,"about_ca_system_score_codex":0.0014539508,"about_ca_system_score_gemma":0.0067640017,"threshold_uncertainty_score":0.18763596},"labels":[],"label_agreement":null},{"id":"W152717974","doi":"10.63317/5h9g23tpswhr","title":"Automatically Identifying Changes in the Semantic Orientation of Words","year":2010,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Task (project management); Natural language processing; Word (group theory); Artificial intelligence; Orientation (vector space); Semantic change; Meaning (existential); Baseline (sea); Word-sense disambiguation; Semantic role labeling; Linguistics; Psychology; Mathematics","score_opus":0.016401369174372914,"score_gpt":0.3180775501979433,"score_spread":0.3016761810235704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W152717974","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9037272,0.0016062787,0.07369514,0.00044381307,0.00028909335,0.00027094738,0.0071623013,0.004867453,0.007937814],"genre_scores_gemma":[0.89154214,0.00048345796,0.09284019,0.00015258341,0.00011115044,0.00016817935,0.011845863,0.00035356497,0.0025029213],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99857605,0.0001821846,0.00018977908,0.00065862684,0.00026806793,0.00012533754],"domain_scores_gemma":[0.9966137,0.0010574366,0.0007312804,0.0003810396,0.0010540907,0.00016254358],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00087622134,0.00075647933,0.0005716384,0.004839412,0.0005432747,0.0020694702,0.0006447171,0.0008537995,0.0014693409],"category_scores_gemma":[0.006003547,0.00039479556,0.0006233265,0.0032245587,0.00073706324,0.0030701754,0.0011726483,0.00094479095,0.001754612],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088964327,0.00036351764,0.15654646,0.0007395264,0.00017684193,0.0006237101,0.0023547949,0.0048835054,0.102080174,0.0028024467,0.014247125,0.7142922],"study_design_scores_gemma":[0.00030454644,0.0009077799,0.48504567,0.0003193175,0.0005319682,0.0026164574,0.010751839,0.27228183,0.11386312,0.031132111,0.08188752,0.0003578445],"about_ca_topic_score_codex":0.004963121,"about_ca_topic_score_gemma":0.008348058,"teacher_disagreement_score":0.004963121,"about_ca_system_score_codex":0.0008255879,"about_ca_system_score_gemma":0.0008612178,"threshold_uncertainty_score":0.009868503},"labels":[],"label_agreement":null},{"id":"W1532575505","doi":"10.1007/3-540-47922-8_22","title":"Text Summarization as Controlled Search","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Automatic summarization; Computer science; Generator (circuit theory); Reliability (semiconductor); Set (abstract data type); Segmentation; Artificial intelligence; Process (computing); Quality (philosophy); Natural language processing; Information retrieval; Power (physics)","score_opus":0.016158621043454276,"score_gpt":0.27138025263504706,"score_spread":0.2552216315915928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1532575505","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014570665,0.0044763377,0.9424285,0.0007072673,0.00036827603,0.0006858192,0.0015999576,0.014797056,0.020366183],"genre_scores_gemma":[0.23988454,0.0023394781,0.6941118,0.0003667997,0.000601604,0.0007755855,0.006456454,0.0020499674,0.05341374],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982697,0.00070260244,0.00014710391,0.0003147157,0.00048380566,0.000082143444],"domain_scores_gemma":[0.99594694,0.0026021057,0.00025698115,0.0006187593,0.0005026526,0.00007264527],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013190843,0.0009404818,0.001681426,0.0031275707,0.00083410734,0.0024914856,0.0018257317,0.0010192731,0.020444917],"category_scores_gemma":[0.006008851,0.0005677455,0.000911003,0.004098284,0.0007685232,0.003956918,0.0013398654,0.00075316976,0.006805085],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008846699,0.00020486819,0.00027364478,0.0013260522,0.00014854615,0.00022801514,0.00045047738,0.012028624,0.042734895,0.06251096,0.045681577,0.8335276],"study_design_scores_gemma":[0.0003869783,0.00090852793,0.0012289528,0.00026013804,0.00050643296,0.00093118026,0.00044686973,0.47597066,0.09343453,0.26549459,0.16024402,0.00018716555],"about_ca_topic_score_codex":0.0012758856,"about_ca_topic_score_gemma":0.0015803056,"teacher_disagreement_score":0.020444917,"about_ca_system_score_codex":0.0005749558,"about_ca_system_score_gemma":0.00073584035,"threshold_uncertainty_score":0.06839508},"labels":[],"label_agreement":null},{"id":"W1538612309","doi":"10.1007/978-3-540-88808-6_5","title":"Connecting Legacy Code, Business Rules and Documentation","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; USable; Documentation; Legacy system; Source code; Reverse engineering; Legacy code; Software engineering; Code (set theory); Business rule; Semantics of Business Vocabulary and Business Rules; Internal documentation; KPI-driven code analysis; Programming language; Business process; World Wide Web; Static program analysis; Software development; Software; Set (abstract data type); Engineering","score_opus":0.016064652009727676,"score_gpt":0.2751270604093387,"score_spread":0.25906240839961103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1538612309","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027678736,0.009709485,0.6917524,0.010390652,0.00086408237,0.00011738498,0.0008919072,0.005515693,0.25307956],"genre_scores_gemma":[0.2936633,0.022667006,0.51213884,0.0024779881,0.0009970743,0.00018852574,0.004068693,0.0039816806,0.1598169],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.999076,0.00022320123,0.00007990799,0.00014104499,0.00042001894,0.000059691574],"domain_scores_gemma":[0.9950127,0.00294241,0.00025328767,0.0009852762,0.0006702283,0.00013623094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013251366,0.0005096729,0.00038168777,0.003682233,0.00090902485,0.00638856,0.0013018642,0.0015795693,0.012328248],"category_scores_gemma":[0.013633655,0.00093699765,0.000386916,0.0050834967,0.0018114314,0.01242972,0.003463409,0.0025242823,0.004059299],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030372421,0.000056527322,0.0020714495,0.00047487448,0.000028880224,0.00049082155,0.0025378973,0.00242496,0.0021277284,0.3652257,0.029806877,0.59472394],"study_design_scores_gemma":[0.000011087446,0.00002827857,0.0018660538,0.0007418945,0.000045652814,0.00092327833,0.001001156,0.014117092,0.0028342586,0.6290436,0.34934726,0.000040445415],"about_ca_topic_score_codex":0.0028305883,"about_ca_topic_score_gemma":0.003484439,"teacher_disagreement_score":0.012328248,"about_ca_system_score_codex":0.00092712167,"about_ca_system_score_gemma":0.0014109147,"threshold_uncertainty_score":0.041242063},"labels":[],"label_agreement":null},{"id":"W1544240449","doi":"10.1007/3-540-45486-1_4","title":"Using Noun Phrase Heads to Extract Document Keyphrases","year":2000,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":236,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Noun phrase; Automatic summarization; Natural language processing; Artificial intelligence; Phrase; Task (project management); Extractor; Head (geology); Noun; Proper noun; Information retrieval; Linguistics","score_opus":0.025876714810008236,"score_gpt":0.31183386811917013,"score_spread":0.2859571533091619,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1544240449","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06625529,0.0034909656,0.84257656,0.0005862986,0.00085359416,0.0009565554,0.017655872,0.052351974,0.015272913],"genre_scores_gemma":[0.14528714,0.0028402237,0.80033666,0.00022407279,0.0004122438,0.0004819517,0.028722776,0.00375346,0.017941535],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99933475,0.00005795266,0.000086231616,0.00020134854,0.00023721352,0.00008243225],"domain_scores_gemma":[0.9964753,0.0014048143,0.00031023961,0.00030165733,0.0013566046,0.00015135843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070029055,0.0019519504,0.0011969587,0.0060499785,0.0010443765,0.0027988807,0.0007317809,0.0010575624,0.011895568],"category_scores_gemma":[0.0036693073,0.0008113304,0.0010602408,0.004638606,0.000534061,0.003023813,0.0012792886,0.0013261448,0.021410199],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063154224,0.00011223633,0.002544596,0.0017350015,0.00009180542,0.0011761172,0.0008391306,0.0008613307,0.20140679,0.004592314,0.02451679,0.76149243],"study_design_scores_gemma":[0.00030181298,0.00095262524,0.026042748,0.00064731296,0.0008528254,0.0052169473,0.0034621167,0.108234055,0.56359565,0.029953659,0.26029733,0.00044288937],"about_ca_topic_score_codex":0.002821813,"about_ca_topic_score_gemma":0.0039766436,"teacher_disagreement_score":0.011895568,"about_ca_system_score_codex":0.00064197567,"about_ca_system_score_gemma":0.0015433516,"threshold_uncertainty_score":0.039794624},"labels":[],"label_agreement":null},{"id":"W1546953651","doi":"10.1007/11581062_61","title":"Automatic Keyword Extraction by Server Log Analysis","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Web page; Information retrieval; Static web page; World Wide Web; Web server; Web navigation; The Internet","score_opus":0.011167238721283796,"score_gpt":0.27492132079478504,"score_spread":0.26375408207350126,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1546953651","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.075584635,0.0035394034,0.7895546,0.0005223937,0.00041320722,0.00049219973,0.01240016,0.10837224,0.009121183],"genre_scores_gemma":[0.34761077,0.0033527652,0.57453144,0.00027260403,0.0003829999,0.0004405843,0.035393387,0.0051324638,0.032883074],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99916315,0.000082773266,0.00010145917,0.00013020087,0.00043479586,0.00008765115],"domain_scores_gemma":[0.9978956,0.0007847112,0.00015815243,0.00032705255,0.00075785856,0.0000766812],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00042770032,0.0011781125,0.0015283265,0.004703983,0.000619342,0.0019526697,0.0010022054,0.00062781124,0.007658174],"category_scores_gemma":[0.0025862087,0.00056769943,0.00079351966,0.004208406,0.00026009095,0.002800674,0.0009091356,0.00058624696,0.018203698],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006070975,0.00016901197,0.0030421792,0.0006520151,0.00005995244,0.00046260768,0.00012983115,0.0025639813,0.13420203,0.0015166384,0.034782205,0.82181245],"study_design_scores_gemma":[0.00017141442,0.0003967885,0.013021241,0.00016491777,0.00038322964,0.004369494,0.0005485669,0.40827298,0.4629327,0.015986856,0.0935556,0.0001963045],"about_ca_topic_score_codex":0.0018958802,"about_ca_topic_score_gemma":0.0028525533,"teacher_disagreement_score":0.007658174,"about_ca_system_score_codex":0.0004035981,"about_ca_system_score_gemma":0.0011879151,"threshold_uncertainty_score":0.02561915},"labels":[],"label_agreement":null},{"id":"W1578895891","doi":"10.2224/sbp.2009.37.2.145","title":"Public recognition of major works in psychology: Rise and fall over time","year":2009,"lang":"en","type":"article","venue":"Social Behavior and Personality An International Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Bishop's University","funders":"","keywords":"Psychology; Period (music); History of psychology; Cognition; Psychoanalysis; Social psychology; Social science; Sociology; Psychiatry; Philosophy; Aesthetics","score_opus":0.06413617509517439,"score_gpt":0.3821694435155543,"score_spread":0.31803326842037993,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1578895891","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9365853,0.01787834,0.0006710275,0.009231434,0.00058225374,0.000044957436,0.01653388,0.00036328527,0.018109478],"genre_scores_gemma":[0.9841381,0.0027698828,0.0002921332,0.00034128758,0.0007253786,0.000038071365,0.0075845993,0.000078088095,0.004032524],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99451387,0.0005345034,0.00097560784,0.0007744948,0.00246771,0.0007337753],"domain_scores_gemma":[0.7600456,0.08653291,0.10283395,0.0076054432,0.030528655,0.012453497],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.004993063,0.00023976916,0.0008380689,0.017179027,0.001257504,0.0043267766,0.0010722612,0.001777569,0.010604365],"category_scores_gemma":[0.06980259,0.00026387506,0.00059931277,0.021746004,0.0015468864,0.005290731,0.0036148794,0.0026716932,0.003267718],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008222833,0.00015907695,0.8943671,0.0011188808,0.0002013078,0.00042028335,0.006391713,0.00017290382,0.0009903748,0.002336695,0.01587753,0.07714187],"study_design_scores_gemma":[0.0000112671805,0.00007388941,0.98482543,0.00013852335,0.000039350372,0.00027612757,0.0028692868,0.00014994339,0.00033484423,0.0005525178,0.010704357,0.000024545068],"about_ca_topic_score_codex":0.0054193614,"about_ca_topic_score_gemma":0.0072645624,"teacher_disagreement_score":0.9950069,"about_ca_system_score_codex":0.001603887,"about_ca_system_score_gemma":0.0012700714,"threshold_uncertainty_score":0.035475075},"labels":[],"label_agreement":null},{"id":"W1593294079","doi":"10.1007/11424918_33","title":"A Document Browsing Tool: Using Lexical Classes to Convey Information","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Digital library; Field (mathematics); Information retrieval; World Wide Web; Linguistics","score_opus":0.019211746900202824,"score_gpt":0.29090359449865594,"score_spread":0.2716918475984531,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1593294079","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020622317,0.0010621373,0.82814664,0.0006959346,0.00032401792,0.0002812154,0.00755731,0.12230306,0.019007375],"genre_scores_gemma":[0.1211603,0.0014141611,0.8151801,0.0007805964,0.00023927526,0.0005170386,0.011404336,0.018422488,0.030881772],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996835,0.000051011615,0.00004620309,0.000078671415,0.00011337785,0.000027253871],"domain_scores_gemma":[0.9973465,0.0017793556,0.00013342494,0.0002931364,0.00025696878,0.00019058856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006820494,0.0012270316,0.0009228003,0.0036742578,0.00084700505,0.0034673973,0.0014038326,0.0013454651,0.02520404],"category_scores_gemma":[0.003928807,0.0006433053,0.0006252388,0.0026539278,0.0005770626,0.005354751,0.0020346872,0.0014447882,0.009345377],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070278003,0.00018203753,0.0014269982,0.0009132239,0.000058033624,0.0005743921,0.0019543632,0.0009507489,0.04599346,0.024965731,0.14565344,0.77662474],"study_design_scores_gemma":[0.00048751774,0.00033000787,0.0039281845,0.0010259935,0.00033372472,0.002697434,0.0013648138,0.093301155,0.10833787,0.063739605,0.72411245,0.00034122652],"about_ca_topic_score_codex":0.0021740848,"about_ca_topic_score_gemma":0.003089229,"teacher_disagreement_score":0.02520404,"about_ca_system_score_codex":0.0004721124,"about_ca_system_score_gemma":0.00057660654,"threshold_uncertainty_score":0.084315896},"labels":[],"label_agreement":null},{"id":"W1598898694","doi":"","title":"Can Human Assistance Improve a Computational Poet","year":2015,"lang":"en","type":"article","venue":"Proceedings of Bridges 2015: Mathematics, Music, Art, Architecture, Culture","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Phrase; Metric (unit); Computer science; Poetry; Set (abstract data type); Artificial intelligence; Natural language processing; Linguistics; Engineering","score_opus":0.022145023727516006,"score_gpt":0.2754080413410924,"score_spread":0.2532630176135764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1598898694","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39365682,0.0027073505,0.39047942,0.015045622,0.001189367,0.0008563911,0.0022869746,0.015073233,0.17870475],"genre_scores_gemma":[0.79414994,0.00067829,0.1917706,0.0013293932,0.00022275413,0.00021501101,0.001919509,0.00084276195,0.008871861],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9895001,0.0064553255,0.00037779994,0.0014103857,0.0019373922,0.0003190644],"domain_scores_gemma":[0.9470645,0.034134336,0.0026148725,0.009050454,0.006046149,0.0010896397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00925358,0.0012494297,0.0006878305,0.001971223,0.0010282021,0.0066598146,0.0013476106,0.0020028024,0.017858336],"category_scores_gemma":[0.10698168,0.00033195107,0.0005533772,0.0012718241,0.002065187,0.008623388,0.0028323291,0.0015219917,0.010547349],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020693312,0.00071467744,0.023584476,0.0016583678,0.00016861844,0.0002513733,0.005958486,0.016182035,0.015219061,0.042090684,0.05469471,0.83740824],"study_design_scores_gemma":[0.00041084003,0.0022335432,0.03594688,0.001093807,0.00023447647,0.0013308747,0.011669578,0.27959046,0.027224828,0.25422353,0.38561743,0.0004237689],"about_ca_topic_score_codex":0.0017798278,"about_ca_topic_score_gemma":0.0025619771,"teacher_disagreement_score":0.017858336,"about_ca_system_score_codex":0.0009964007,"about_ca_system_score_gemma":0.0015639315,"threshold_uncertainty_score":0.059742033},"labels":[],"label_agreement":null},{"id":"W166693255","doi":"10.4018/978-1-60566-172-8.ch009","title":"A Model for Estimating the Savings from Dimensional vs. Keyword Search","year":2009,"lang":"en","type":"book-chapter","venue":"Advances in database research (ADR) book series/Advances in database research series","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Zipf's law; Computer science; Metadata; Keyword search; Process (computing); Information retrieval; Search cost; Search engine; Keyword density; Data mining; Data science; World Wide Web; Economics; Statistics; Mathematics; Microeconomics","score_opus":0.08805243973955451,"score_gpt":0.41779534693456,"score_spread":0.3297429071950055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W166693255","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10278782,0.0009735459,0.8723003,0.0019506238,0.000053526746,0.00032305458,0.0014135872,0.00091541326,0.019282002],"genre_scores_gemma":[0.69094527,0.0014542226,0.2870992,0.0004380969,0.00007883176,0.00091311854,0.0015300829,0.0003683816,0.017172769],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977246,0.0009261548,0.00012911807,0.0004110536,0.0005903229,0.00021872464],"domain_scores_gemma":[0.97490543,0.020890797,0.0016676404,0.0010910443,0.0012072026,0.00023797722],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043201502,0.0008903442,0.0013177578,0.0020215039,0.0005932573,0.0031783658,0.0027509402,0.0024671142,0.00946746],"category_scores_gemma":[0.031254638,0.00086861895,0.0011121803,0.0038303176,0.0012964458,0.007668726,0.0012516398,0.0017393337,0.002552734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028094815,0.00012054689,0.0040814476,0.00018797607,0.000067297784,0.00013109023,0.0001703931,0.8232542,0.0014099686,0.117951386,0.0033559566,0.048988733],"study_design_scores_gemma":[0.000019222676,0.000065607484,0.0007899153,0.000020012645,0.000031223928,0.00013354258,0.000059094647,0.9592166,0.00040200783,0.038008463,0.0012220682,0.000032228447],"about_ca_topic_score_codex":0.008535706,"about_ca_topic_score_gemma":0.0047736387,"teacher_disagreement_score":0.00946746,"about_ca_system_score_codex":0.0037678296,"about_ca_system_score_gemma":0.0017236933,"threshold_uncertainty_score":0.031671762},"labels":[],"label_agreement":null},{"id":"W1728627389","doi":"","title":"Design and Considerations of a Searching Software","year":2015,"lang":"en","type":"article","venue":"Transactions on machine learning and data mining","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"Order (exchange); Computer science; Software; Data science; Business; Finance","score_opus":0.0924846284533971,"score_gpt":0.35050625767348115,"score_spread":0.25802162922008404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1728627389","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014507608,0.00018898537,0.9497103,0.0011403447,0.0001311664,0.001603236,0.00039283835,0.022768235,0.009557221],"genre_scores_gemma":[0.09837124,0.00019853108,0.8856763,0.00074887346,0.00006676056,0.0016224704,0.00081374304,0.002446157,0.010055968],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9941962,0.0018453287,0.00071951014,0.00095424877,0.0018574797,0.0004272301],"domain_scores_gemma":[0.98849815,0.0050494033,0.00032590723,0.0020894206,0.0033916067,0.00064542494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0058600553,0.0008431351,0.0010588014,0.0018782413,0.0010798367,0.00474429,0.004227924,0.0024773863,0.014407239],"category_scores_gemma":[0.020415805,0.0010196252,0.001157595,0.0014379963,0.000929011,0.0049377135,0.0016083026,0.0015151551,0.008103084],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036998647,0.0009485425,0.011812229,0.001843577,0.0003700555,0.0017822621,0.0016720057,0.01592483,0.10331584,0.088185094,0.039707433,0.7307383],"study_design_scores_gemma":[0.0014000018,0.0031738502,0.0064783785,0.0007228073,0.0006857043,0.0036392917,0.0013718958,0.5225609,0.1591161,0.05890294,0.24162364,0.00032458935],"about_ca_topic_score_codex":0.0021298868,"about_ca_topic_score_gemma":0.0015399107,"teacher_disagreement_score":0.014407239,"about_ca_system_score_codex":0.0007147793,"about_ca_system_score_gemma":0.002165154,"threshold_uncertainty_score":0.04819703},"labels":[],"label_agreement":null},{"id":"W1777978449","doi":"","title":"Topical Segmentation: a Study of Human Performance and a New Measure of Quality.","year":2012,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Metric (unit); Segmentation; Computer science; Measure (data warehouse); Quality (philosophy); Simple (philosophy); Scale (ratio); Agreement; Artificial intelligence; Natural language processing; Machine learning; Data mining; Linguistics; Epistemology","score_opus":0.07960152078766151,"score_gpt":0.3763586674321693,"score_spread":0.29675714664450775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1777978449","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.74025893,0.010099272,0.2254902,0.0005896454,0.0004116259,0.00047372608,0.0025418748,0.0024565384,0.017678184],"genre_scores_gemma":[0.9492123,0.00080156524,0.045513954,0.00009687999,0.00013334319,0.00024251189,0.0014636264,0.0005197289,0.0020161294],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9830998,0.007084608,0.0012850455,0.0041496065,0.003969957,0.00041090566],"domain_scores_gemma":[0.8617039,0.09424691,0.015848318,0.008929744,0.016863419,0.0024077268],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018640973,0.0008614972,0.0009102742,0.0049862424,0.0015217693,0.0030748465,0.0011279129,0.0016274134,0.0019204672],"category_scores_gemma":[0.09381014,0.00045036303,0.0007318196,0.005812716,0.0020355405,0.004868217,0.002267978,0.0008034747,0.0011080886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0050700773,0.00038866888,0.28473493,0.004213261,0.0018929305,0.0006830184,0.035385877,0.0116000455,0.12538889,0.005339986,0.014935791,0.51036656],"study_design_scores_gemma":[0.0002543482,0.0036833982,0.73532754,0.00053222786,0.0008474967,0.0034148954,0.013089996,0.14625458,0.052578352,0.016988413,0.02636993,0.000658775],"about_ca_topic_score_codex":0.0034111235,"about_ca_topic_score_gemma":0.0055005215,"teacher_disagreement_score":0.018640973,"about_ca_system_score_codex":0.0010006703,"about_ca_system_score_gemma":0.0005218544,"threshold_uncertainty_score":0.098584116},"labels":[],"label_agreement":null},{"id":"W178949263","doi":"10.1007/978-3-319-12024-9_14","title":"Semantic Facets for Scientific Information Retrieval","year":2014,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Information retrieval; Sentence; Ontology; Semantic search; Semantics (computer science); Filter (signal processing); Natural language processing; World Wide Web; Semantic Web","score_opus":0.03733782229050642,"score_gpt":0.3114533557815188,"score_spread":0.27411553349101236,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W178949263","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008939302,0.04718627,0.77810913,0.0041519455,0.0025074142,0.00035955707,0.0031502673,0.0047671073,0.15082905],"genre_scores_gemma":[0.14418605,0.053950723,0.6704505,0.0015469716,0.0023608226,0.0004900278,0.010109124,0.0016453363,0.11526047],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995049,0.00010534376,0.000058676596,0.000071906055,0.00021972114,0.00003949453],"domain_scores_gemma":[0.9994492,0.00020865696,0.000030582098,0.00015188994,0.00012584002,0.000033813325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070315844,0.00075954717,0.00073846796,0.0024911882,0.00076836906,0.004601139,0.0009058836,0.00071948604,0.011018657],"category_scores_gemma":[0.0025949983,0.00039581454,0.0008427544,0.0045472877,0.001313177,0.007613374,0.0018352389,0.0015575502,0.0046927095],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043188786,0.00002659465,0.00023591335,0.0006283719,0.000029756713,0.0000810697,0.0003997881,0.001618346,0.002431887,0.5905765,0.057986356,0.34594214],"study_design_scores_gemma":[0.0000074977975,0.000015441661,0.00030313956,0.00026631824,0.00003133505,0.00024636165,0.00026274918,0.0140051935,0.0019406928,0.66692436,0.31597346,0.00002345227],"about_ca_topic_score_codex":0.0033348396,"about_ca_topic_score_gemma":0.0036689742,"teacher_disagreement_score":0.011018657,"about_ca_system_score_codex":0.0012777317,"about_ca_system_score_gemma":0.0010968246,"threshold_uncertainty_score":0.03686112},"labels":[],"label_agreement":null},{"id":"W1809680580","doi":"10.1111/j.1365-2575.2010.00368.x","title":"Using decision tree modelling to support Peircian abduction in IS research: a systematic approach for generating and evaluating hypotheses for systematic theory development","year":2011,"lang":"en","type":"article","venue":"Information Systems Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"Virginia Commonwealth University","keywords":"Computer science; Data science; Management science; Development (topology); Empirical research; Decision tree; Development theory; Tree (set theory); Knowledge management; Data mining; Epistemology; Mathematics; Engineering","score_opus":0.46641362650993345,"score_gpt":0.4305231638408666,"score_spread":0.03589046266906687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1809680580","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012886699,0.00022559313,0.98400056,0.00079607824,0.000021793423,0.00050850585,0.00013100081,0.00016606979,0.0012636014],"genre_scores_gemma":[0.09216371,0.0002548549,0.90642834,0.00009349623,0.0000175631,0.0006467571,0.0001965036,0.000015397634,0.00018336267],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9639976,0.027734354,0.0022610342,0.0012930429,0.0043555074,0.00035852392],"domain_scores_gemma":[0.83324605,0.14791903,0.0059678173,0.006430032,0.005774847,0.00066225213],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03786034,0.001578431,0.0016143442,0.0072596977,0.0016784725,0.0046638097,0.0035040413,0.002356151,0.0023429764],"category_scores_gemma":[0.09433106,0.0011023291,0.0023411356,0.0051579564,0.0037585779,0.008026021,0.003282359,0.0031155888,0.00039623733],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005198928,0.0009360355,0.013077012,0.0017057777,0.0005111236,0.0006616097,0.0077490215,0.2261297,0.0043459884,0.40545154,0.0019799522,0.33693245],"study_design_scores_gemma":[0.00018176991,0.00030353328,0.00092088256,0.00062903005,0.00014128174,0.00015048787,0.0006447407,0.6078163,0.002885771,0.3811313,0.005095807,0.00009910717],"about_ca_topic_score_codex":0.0034359954,"about_ca_topic_score_gemma":0.0073579084,"teacher_disagreement_score":0.03786034,"about_ca_system_score_codex":0.002564547,"about_ca_system_score_gemma":0.005574398,"threshold_uncertainty_score":0.20022708},"labels":[],"label_agreement":null},{"id":"W18253573","doi":"","title":"Linked Opinions: Describing Sentiments on the Structured Web of Data.","year":2011,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Semantic Web; World Wide Web; Sentiment analysis; Ontology; Data science; Process (computing); Publishing; Information retrieval; Artificial intelligence; Political science; Epistemology","score_opus":0.16491186477849681,"score_gpt":0.32247194816388364,"score_spread":0.15756008338538682,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W18253573","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51073897,0.005758139,0.283741,0.016423281,0.0014335976,0.0020167325,0.08808994,0.009871558,0.08192682],"genre_scores_gemma":[0.8248482,0.001857012,0.12632462,0.0009883725,0.00032126874,0.0009481244,0.034093916,0.00086154003,0.009756976],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998038,0.0011050997,0.00020204071,0.00014842639,0.0004467685,0.000059783753],"domain_scores_gemma":[0.9901128,0.007924754,0.0006247788,0.00036506713,0.0006981113,0.00027445928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002762125,0.00049206155,0.00020698436,0.0035223456,0.001033746,0.0027019354,0.00042316873,0.0007797572,0.0049370504],"category_scores_gemma":[0.017703721,0.00016132228,0.0003617323,0.003970923,0.0007042519,0.00576368,0.001999395,0.0007661311,0.0011943949],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010983044,0.00024020362,0.030581497,0.004623723,0.00034027058,0.0035152812,0.1535728,0.0049233194,0.024298282,0.06336224,0.19149165,0.5219524],"study_design_scores_gemma":[0.00010778366,0.0002723726,0.03953628,0.0017293179,0.00014972729,0.0016438244,0.11986629,0.094607994,0.0068094516,0.14011237,0.5949531,0.0002115692],"about_ca_topic_score_codex":0.0017938679,"about_ca_topic_score_gemma":0.005899365,"teacher_disagreement_score":0.0049370504,"about_ca_system_score_codex":0.0008028219,"about_ca_system_score_gemma":0.00049339497,"threshold_uncertainty_score":0.01651609},"labels":[],"label_agreement":null},{"id":"W1910591194","doi":"10.1109/fuzzy.1997.619736","title":"Elicitation of membership functions: how far can theory take us?","year":2002,"lang":"en","type":"article","venue":"Proceedings of 6th International Fuzzy Systems Conference","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Management science; Cognitive science; Psychology; Engineering","score_opus":0.03719236577059323,"score_gpt":0.2486746373106849,"score_spread":0.21148227154009167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1910591194","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009650738,0.11175931,0.41566584,0.41213587,0.0035496473,0.00026437544,0.0004899185,0.0007832677,0.045701],"genre_scores_gemma":[0.40105033,0.10685503,0.422558,0.05459836,0.004999473,0.0017383546,0.0009525609,0.00061275973,0.006635171],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.95732063,0.028187828,0.0022426108,0.002544754,0.008576981,0.0011271383],"domain_scores_gemma":[0.8959617,0.06856537,0.0042945803,0.012017291,0.016700866,0.0024602409],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07501705,0.0036313944,0.0068672174,0.006422973,0.004415709,0.02161795,0.009127845,0.013509692,0.009633452],"category_scores_gemma":[0.11895395,0.0014837804,0.0025431884,0.006311562,0.029853377,0.049276758,0.0073710815,0.016141418,0.004962101],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016097282,0.00015839798,0.00093952706,0.0025261985,0.00021873937,0.00012716037,0.002335318,0.0034876002,0.0003551923,0.86573404,0.015468674,0.108488165],"study_design_scores_gemma":[0.000034920075,0.00006720034,0.00019569501,0.0020577917,0.00003549892,0.000049253038,0.0018757924,0.004722555,0.00026811418,0.9713946,0.019207925,0.00009074507],"about_ca_topic_score_codex":0.0070325965,"about_ca_topic_score_gemma":0.004678395,"teacher_disagreement_score":0.07501705,"about_ca_system_score_codex":0.0098550245,"about_ca_system_score_gemma":0.00894918,"threshold_uncertainty_score":0.3967328},"labels":[],"label_agreement":null},{"id":"W193970460","doi":"10.1007/978-3-642-34752-8_11","title":"Improving Supervised Keyphrase Indexer Classification of Keyphrases with Text Denoising","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Information retrieval; Pattern recognition (psychology)","score_opus":0.018745757084550643,"score_gpt":0.25325356572081503,"score_spread":0.2345078086362644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W193970460","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.098370984,0.0033905914,0.8795751,0.00040206956,0.00057907135,0.00020315647,0.0013969,0.012555279,0.0035268697],"genre_scores_gemma":[0.36684704,0.0019018253,0.5967061,0.00043845558,0.0008967751,0.00021606818,0.007325082,0.0011218362,0.024546858],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986922,0.0001823233,0.00009688669,0.00035849924,0.0005017993,0.00016828443],"domain_scores_gemma":[0.9972573,0.0008886777,0.00021543526,0.0005674291,0.00092914567,0.0001420403],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013466361,0.0015352111,0.0022555506,0.003147478,0.0005256835,0.0016449607,0.0013247234,0.0016233232,0.003686516],"category_scores_gemma":[0.0040118536,0.00036032838,0.0012137301,0.002215907,0.00045951433,0.0020950623,0.0012364144,0.0016774038,0.008272681],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009350859,0.00024860742,0.0013505941,0.00020406037,0.00008249433,0.0001137253,0.000054750366,0.005381874,0.12790497,0.00096679555,0.010070374,0.8526866],"study_design_scores_gemma":[0.000069391106,0.00033358895,0.0039176825,0.00003240599,0.00014648814,0.0004420126,0.00011570628,0.8803622,0.102297954,0.002876482,0.0093585625,0.0000475781],"about_ca_topic_score_codex":0.001974928,"about_ca_topic_score_gemma":0.0033999006,"teacher_disagreement_score":0.003686516,"about_ca_system_score_codex":0.00041711333,"about_ca_system_score_gemma":0.0008054437,"threshold_uncertainty_score":0.012332618},"labels":[],"label_agreement":null},{"id":"W1966427115","doi":"10.1145/2232817.2232866","title":"Investigating keyphrase indexing with text denoising","year":2012,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Search engine indexing; Computer science; Benchmark (surveying); Noise reduction; Artificial intelligence; Natural language processing; Noise (video); Energy (signal processing); Pattern recognition (psychology); Information retrieval; Image (mathematics); Mathematics","score_opus":0.01774092605419382,"score_gpt":0.26945065966914467,"score_spread":0.25170973361495086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1966427115","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5229497,0.013543648,0.40993956,0.0012899197,0.000998739,0.0008614235,0.0036603236,0.03153271,0.015224],"genre_scores_gemma":[0.5098613,0.0024357114,0.45743862,0.00053778145,0.0006503186,0.00044762777,0.015527346,0.0012964396,0.011804798],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99580514,0.0010882411,0.0004186572,0.0009478551,0.0014590591,0.0002809599],"domain_scores_gemma":[0.98394644,0.009824095,0.000983668,0.0028524739,0.0020119043,0.0003814384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006485353,0.0016784142,0.0018954797,0.004473083,0.0009816521,0.0029018272,0.0020786023,0.0017431978,0.0027240796],"category_scores_gemma":[0.029142214,0.00042164788,0.0010886203,0.0041048657,0.001601915,0.0070392257,0.0020706002,0.0018607359,0.0035393692],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019633987,0.0006784911,0.006596769,0.0018945944,0.0003462993,0.00032992373,0.0010613744,0.027444836,0.13759021,0.0038888531,0.01195332,0.806252],"study_design_scores_gemma":[0.00026674487,0.0022845597,0.013457549,0.00013387062,0.0003584833,0.0012739127,0.0012645053,0.5637904,0.37118533,0.007937811,0.037794605,0.0002522784],"about_ca_topic_score_codex":0.004195049,"about_ca_topic_score_gemma":0.004908188,"teacher_disagreement_score":0.006485353,"about_ca_system_score_codex":0.001022721,"about_ca_system_score_gemma":0.0011954785,"threshold_uncertainty_score":0.03429824},"labels":[],"label_agreement":null},{"id":"W1967235662","doi":"10.1038/nmeth.2490","title":"Plotting symbols","year":2013,"lang":"en","type":"article","venue":"Nature Methods","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada's Michael Smith Genome Sciences Centre","funders":"","keywords":"Ambiguity; Computer science; Computational biology; Artificial intelligence; Biology; Programming language","score_opus":0.009798851858909177,"score_gpt":0.39838186598130454,"score_spread":0.3885830141223954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1967235662","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008108988,0.00096337777,0.42622724,0.0031940069,0.008452573,0.0006318263,0.05496766,0.1690097,0.3284447],"genre_scores_gemma":[0.11079346,0.0019408461,0.4284322,0.0019931796,0.0016999632,0.0021387283,0.053221803,0.07699943,0.32278046],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99888486,0.00029277464,0.00008296776,0.0002926917,0.00034128947,0.000105462605],"domain_scores_gemma":[0.9926109,0.0030397663,0.00040041332,0.0016172692,0.0020220003,0.0003097225],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0008159625,0.002071151,0.00080763485,0.004642441,0.0012778747,0.0040804707,0.0016440996,0.0013320969,0.3765165],"category_scores_gemma":[0.017383367,0.00058724795,0.00088228984,0.005223702,0.00091032276,0.0031195085,0.0021316055,0.0023614888,0.17835379],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039944,0.00006199827,0.00090830785,0.0006144828,0.000022266971,0.00019552329,0.00075525546,0.0019422387,0.005458513,0.07402399,0.6768392,0.23877889],"study_design_scores_gemma":[0.00007866879,0.000049837578,0.00082511397,0.00019386334,0.000022373824,0.00018033742,0.00022081226,0.0051566553,0.0075949696,0.032722913,0.95290625,0.00004823643],"about_ca_topic_score_codex":0.001899303,"about_ca_topic_score_gemma":0.0014559022,"teacher_disagreement_score":0.3765165,"about_ca_system_score_codex":0.0008464838,"about_ca_system_score_gemma":0.0013787807,"threshold_uncertainty_score":0.88932353},"labels":[],"label_agreement":null},{"id":"W1968731131","doi":"10.1145/2505515.2507854","title":"Modeling latent topic interactions using quantum interference for information retrieval","year":2013,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinterpretation; Probabilistic logic; Computer science; Quantum; Quantum interference; Interference (communication); Information retrieval; Theoretical computer science; Artificial intelligence; Quantum mechanics; Physics; Telecommunications","score_opus":0.05218644465248692,"score_gpt":0.31901374649699377,"score_spread":0.26682730184450687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1968731131","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04947016,0.0019536563,0.94264495,0.0014294249,0.00014134441,0.0000760559,0.00020757822,0.00028853156,0.0037882673],"genre_scores_gemma":[0.87295365,0.0025726133,0.11552856,0.00057216006,0.0008040932,0.00033403625,0.0005385083,0.0001579518,0.0065384717],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99746275,0.0015000707,0.00011302429,0.00029403967,0.00038203257,0.00024803943],"domain_scores_gemma":[0.9901245,0.008390996,0.0004828072,0.0004973744,0.0003216829,0.00018272619],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041224356,0.0006777146,0.0018047969,0.0016647031,0.001008382,0.0026347295,0.0022518323,0.0022396722,0.0033147694],"category_scores_gemma":[0.014952892,0.0006454204,0.0016185466,0.0022312393,0.0018594753,0.0047305305,0.002044345,0.0024145907,0.00067679776],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067570235,0.00042829552,0.003383568,0.0005789441,0.00053973513,0.00046553774,0.0011727879,0.3244804,0.010239839,0.58547515,0.0047924826,0.067767486],"study_design_scores_gemma":[0.000039213373,0.00004200983,0.00032226846,0.000011543915,0.00005266671,0.0000461983,0.00003053376,0.8554794,0.00046462633,0.14267667,0.00080913195,0.000025642237],"about_ca_topic_score_codex":0.003694069,"about_ca_topic_score_gemma":0.0033853545,"teacher_disagreement_score":0.0041224356,"about_ca_system_score_codex":0.0010731836,"about_ca_system_score_gemma":0.001173323,"threshold_uncertainty_score":0.02180177},"labels":[],"label_agreement":null},{"id":"W1969838600","doi":"10.3166/isi.11.4.81-97","title":"Réduction de l'espace de recherche par les techniques d'élagage","year":2006,"lang":"fr","type":"article","venue":"Ingénierie des systèmes d information","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Art","score_opus":0.0592299782641279,"score_gpt":0.317243919151696,"score_spread":0.25801394088756807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1969838600","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025418831,0.0037005895,0.9605102,0.0007728123,0.00012906278,0.00027857386,0.00033612328,0.0028327873,0.0060210135],"genre_scores_gemma":[0.09531165,0.0028545293,0.8887027,0.0002369932,0.00023785929,0.00044419323,0.0010638462,0.00097125245,0.010176973],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9888026,0.0038184952,0.0005932945,0.0014180166,0.0049993508,0.0003681749],"domain_scores_gemma":[0.9835596,0.008870199,0.0007810925,0.00423688,0.002307416,0.0002447569],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051762965,0.001459223,0.001735042,0.0045622163,0.0011600726,0.0031589665,0.0016469314,0.0009470135,0.0059424946],"category_scores_gemma":[0.024535526,0.0005653759,0.001729193,0.0038879504,0.0013198691,0.003777123,0.0020940267,0.001805602,0.004677423],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030703386,0.00018314378,0.0019722634,0.0006249665,0.00010748227,0.00011255541,0.00082555873,0.007564584,0.01900215,0.025112774,0.0051676338,0.93901986],"study_design_scores_gemma":[0.0004748074,0.0010195086,0.0147668505,0.0007146724,0.00081222266,0.004515351,0.002406737,0.31700322,0.1442875,0.10844363,0.4052896,0.0002658118],"about_ca_topic_score_codex":0.0030765075,"about_ca_topic_score_gemma":0.0031325729,"teacher_disagreement_score":0.0059424946,"about_ca_system_score_codex":0.000849686,"about_ca_system_score_gemma":0.0019796714,"threshold_uncertainty_score":0.027375162},"labels":[],"label_agreement":null},{"id":"W1970043789","doi":"10.1145/1645953.1646076","title":"Detecting topic evolution in scientific literature","year":2009,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":165,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"National Science Foundation","keywords":"Latent Dirichlet allocation; Computer science; Topic model; Data science; Citation; Inheritance (genetic algorithm); Scientific literature; Information retrieval; World Wide Web","score_opus":0.006675345525235707,"score_gpt":0.262146813315761,"score_spread":0.2554714677905253,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1970043789","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80460423,0.009264443,0.17833808,0.0011140761,0.00018496603,0.00020219278,0.0012482113,0.0011986141,0.0038453317],"genre_scores_gemma":[0.93670446,0.0018840672,0.05739059,0.00011269807,0.00036839786,0.00015280992,0.0019682834,0.000090927126,0.0013278326],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9962012,0.0010584444,0.00038429926,0.0009771591,0.0011292242,0.00024959462],"domain_scores_gemma":[0.96538806,0.022164052,0.005761709,0.0015178641,0.0043996987,0.0007685349],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0077049127,0.0005139778,0.0010388879,0.017419437,0.0012697545,0.0031171606,0.00122204,0.0019993628,0.0006122695],"category_scores_gemma":[0.041957963,0.00045554206,0.0010032731,0.013809923,0.00074179075,0.0051995595,0.0020477585,0.001253953,0.00049979036],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004658366,0.0003323677,0.31858233,0.0010805723,0.0005162393,0.0008286027,0.0056291646,0.03614933,0.020168275,0.016457649,0.008023629,0.5917659],"study_design_scores_gemma":[0.00009140813,0.00031972895,0.27058372,0.00026921922,0.00063922646,0.0023594564,0.0024466182,0.599698,0.02118755,0.079884075,0.02231139,0.00020965036],"about_ca_topic_score_codex":0.0030228372,"about_ca_topic_score_gemma":0.0028568425,"teacher_disagreement_score":0.98258054,"about_ca_system_score_codex":0.0013945637,"about_ca_system_score_gemma":0.0011185894,"threshold_uncertainty_score":0.04074794},"labels":[],"label_agreement":null},{"id":"W1970294714","doi":"10.3115/1614049.1614094","title":"Lycos Retriever","year":2006,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Labrador Retriever; Information retrieval; World Wide Web; Medicine","score_opus":0.0035388341363369272,"score_gpt":0.21494187018193442,"score_spread":0.2114030360455975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1970294714","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019415658,0.004525222,0.18887372,0.0016830958,0.000877566,0.002107167,0.07439089,0.61591345,0.092213295],"genre_scores_gemma":[0.098385364,0.0046132565,0.3962585,0.0031106523,0.0015611806,0.0022482767,0.28607723,0.05289972,0.15484586],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998403,0.00022594028,0.00014836784,0.0003797666,0.000712256,0.0001306544],"domain_scores_gemma":[0.99643815,0.0006994852,0.00022412543,0.00073674734,0.0016513158,0.00025017787],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0015865874,0.0016485732,0.0019375972,0.005126783,0.0011676302,0.0032639476,0.0018773404,0.0012138416,0.05086434],"category_scores_gemma":[0.005910343,0.000547858,0.0007699798,0.0031125778,0.0003590562,0.0037587832,0.0021419616,0.0012571735,0.07997463],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010040851,0.00016559877,0.00187598,0.0013071919,0.000080725564,0.0005734881,0.0006462056,0.00087662006,0.030709215,0.0041509205,0.70043826,0.25817177],"study_design_scores_gemma":[0.00023299897,0.00021414697,0.0027954706,0.00012545701,0.00009181677,0.00090455427,0.00039470682,0.023972057,0.032608766,0.0025081858,0.9360058,0.00014623972],"about_ca_topic_score_codex":0.0049575265,"about_ca_topic_score_gemma":0.006272654,"teacher_disagreement_score":0.94913566,"about_ca_system_score_codex":0.0008219849,"about_ca_system_score_gemma":0.0015572973,"threshold_uncertainty_score":0.17015815},"labels":[],"label_agreement":null},{"id":"W1971022461","doi":"10.1145/1810617.1810648","title":"The impact of resource title on tags in collaborative tagging systems","year":2010,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Folksonomy; Tag system; Computer science; Resource (disambiguation); World Wide Web; Set (abstract data type); Consistency (knowledge bases); Process (computing); Information retrieval","score_opus":0.005319121494490606,"score_gpt":0.2990915330229232,"score_spread":0.2937724115284326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1971022461","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55647755,0.0060142134,0.38077202,0.0035089832,0.0009904576,0.0004936586,0.00068628136,0.00267106,0.04838577],"genre_scores_gemma":[0.95105183,0.0009691891,0.040245842,0.00028636595,0.00026369747,0.000121875906,0.0004356535,0.00061598734,0.0060094674],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.96769494,0.017852232,0.0030004827,0.0025304856,0.00789352,0.0010283982],"domain_scores_gemma":[0.5418033,0.38824114,0.02269416,0.024170453,0.019164015,0.0039268783],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022619648,0.0007974683,0.0011304779,0.0022327441,0.0033861448,0.009275911,0.0013416972,0.0021811852,0.004921929],"category_scores_gemma":[0.1952809,0.0010416494,0.00075345335,0.0028639324,0.0038242748,0.027007436,0.0048142644,0.0025337697,0.002333922],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006523852,0.0010807944,0.21027821,0.0029552812,0.00037179,0.0052330815,0.029322792,0.077222794,0.045090113,0.10376206,0.010847529,0.5073117],"study_design_scores_gemma":[0.00052396505,0.003142888,0.096699566,0.0011945086,0.0017997796,0.010100549,0.014161207,0.5155465,0.11582629,0.163193,0.07665103,0.0011606573],"about_ca_topic_score_codex":0.0029337925,"about_ca_topic_score_gemma":0.0020754545,"teacher_disagreement_score":0.022619648,"about_ca_system_score_codex":0.0027564059,"about_ca_system_score_gemma":0.0017003622,"threshold_uncertainty_score":0.11962557},"labels":[],"label_agreement":null},{"id":"W1977474781","doi":"10.1155/2014/920892","title":"Mental Mechanisms for Topics Identification","year":2014,"lang":"en","type":"article","venue":"Computational Intelligence and Neuroscience","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Computer science; Identification (biology); Quality (philosophy); Statistic; Baseline (sea); Process (computing); Term (time); Focus (optics); Artificial neural network; Artificial intelligence; Cognitive psychology; Psychology; Statistics; Mathematics; Physics","score_opus":0.04112137919388175,"score_gpt":0.3281550001725835,"score_spread":0.2870336209787017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1977474781","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24751614,0.0007804108,0.7292435,0.0019860119,0.0001775298,0.00029872768,0.0006874748,0.0021114773,0.01719869],"genre_scores_gemma":[0.8430872,0.0004553319,0.1500728,0.00033952374,0.00009284364,0.0003484427,0.0007471706,0.00029110603,0.0045656315],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99875236,0.00027106053,0.00009477646,0.00044438688,0.00031279706,0.00012460804],"domain_scores_gemma":[0.9937173,0.0027830696,0.0008368755,0.0015670899,0.0007598692,0.00033568015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019825527,0.00068011606,0.00046458156,0.0018620029,0.00089422317,0.0047264555,0.0018879763,0.0014890601,0.0059033404],"category_scores_gemma":[0.017600851,0.00073923403,0.0015046634,0.00094044144,0.0018593993,0.008543786,0.0023902296,0.0019005057,0.001513101],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010507545,0.0004268139,0.026034264,0.001200109,0.0006288556,0.0006511807,0.008577189,0.048030123,0.15188542,0.41206005,0.008101192,0.341354],"study_design_scores_gemma":[0.00010892029,0.00036854076,0.020229798,0.00015087455,0.00031533005,0.0009982104,0.0015845791,0.41757524,0.05593385,0.48490903,0.017602192,0.0002235091],"about_ca_topic_score_codex":0.0012849498,"about_ca_topic_score_gemma":0.0008185265,"teacher_disagreement_score":0.0059033404,"about_ca_system_score_codex":0.0009983006,"about_ca_system_score_gemma":0.000934268,"threshold_uncertainty_score":0.019748628},"labels":[],"label_agreement":null},{"id":"W1977563360","doi":"10.1145/1088463.1088490","title":"Augmenting conversational dialogue by means of latent semantic googling","year":2005,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Conversation; Semantics (computer science); Search engine indexing; Closeness; Information retrieval; World Wide Web; Probabilistic latent semantic analysis; Natural language processing; Latent semantic analysis; Semantic computing; Artificial intelligence; Semantic Web; Linguistics","score_opus":0.012057681514900432,"score_gpt":0.24926304285076042,"score_spread":0.23720536133585998,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1977563360","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041034814,0.00043391535,0.9436247,0.00040320097,0.000097247495,0.00024823262,0.00069751014,0.007943471,0.005516978],"genre_scores_gemma":[0.44095567,0.00039868787,0.5507807,0.00024813047,0.00014095182,0.00042685837,0.0023660657,0.00061489304,0.004068022],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99746215,0.0012095937,0.00009752882,0.0006055278,0.00047335812,0.00015185375],"domain_scores_gemma":[0.9963295,0.002416725,0.00025588233,0.00053484965,0.00034837166,0.00011473707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021783237,0.0012314563,0.00091206614,0.001990382,0.00097165065,0.002191492,0.0010790116,0.0006842816,0.0044957562],"category_scores_gemma":[0.009631496,0.0004311662,0.0011478082,0.0012370498,0.0010727476,0.0062747854,0.0030306575,0.0012661538,0.002057391],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016735317,0.0005282772,0.005006765,0.001020838,0.00016457196,0.00032038544,0.006545212,0.025214564,0.05165293,0.04402085,0.0057226843,0.8581294],"study_design_scores_gemma":[0.00016171843,0.00045688092,0.0042903847,0.00015370826,0.00021237181,0.00037803594,0.002185834,0.79026777,0.03263786,0.12575078,0.04328178,0.00022280755],"about_ca_topic_score_codex":0.00404414,"about_ca_topic_score_gemma":0.004536101,"teacher_disagreement_score":0.0044957562,"about_ca_system_score_codex":0.0007050272,"about_ca_system_score_gemma":0.0010195067,"threshold_uncertainty_score":0.015039861},"labels":[],"label_agreement":null},{"id":"W1978022428","doi":"10.1109/icdm.2014.120","title":"Mining Contentious Documents Using an Unsupervised Topic Model Based Approach","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Latent Dirichlet allocation; Topic model; Computer science; Viewpoints; Artificial intelligence; Probabilistic logic; Cluster analysis; Expression (computer science); Domain (mathematical analysis); Natural language processing; Information retrieval; Machine learning; Mathematics","score_opus":0.04839903355945884,"score_gpt":0.30069312434877,"score_spread":0.25229409078931114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978022428","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054414425,0.00088373286,0.9410255,0.00028362736,0.00006435302,0.00023467829,0.0005822909,0.0009263844,0.0015850292],"genre_scores_gemma":[0.4963487,0.0007563964,0.49491096,0.0001607613,0.00037195382,0.00061769143,0.00386853,0.00022726804,0.002737777],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99731016,0.0008016726,0.00024172275,0.0007223532,0.00074552343,0.00017860849],"domain_scores_gemma":[0.99536836,0.00289461,0.0004593082,0.00038527022,0.00077838,0.00011411735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002537238,0.0011214915,0.001084888,0.0070834574,0.0009544989,0.0024538846,0.001846538,0.0013888474,0.0008496435],"category_scores_gemma":[0.0072293035,0.000604461,0.0018916163,0.0042568725,0.0006910187,0.0024226683,0.0013549901,0.0013140936,0.00074204383],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086726807,0.00067934126,0.027199242,0.0012255637,0.0010821511,0.0016905458,0.005561105,0.066879,0.061328467,0.03114689,0.0120663615,0.79027414],"study_design_scores_gemma":[0.00008915158,0.00018119681,0.009951571,0.00009476942,0.0003402541,0.0008406285,0.0014532482,0.9222146,0.0144774895,0.03882478,0.011446479,0.00008586865],"about_ca_topic_score_codex":0.0018864529,"about_ca_topic_score_gemma":0.0038342336,"teacher_disagreement_score":0.0070834574,"about_ca_system_score_codex":0.0007375157,"about_ca_system_score_gemma":0.0012591946,"threshold_uncertainty_score":0.013418317},"labels":[],"label_agreement":null},{"id":"W1978265849","doi":"10.1108/00220410610688750","title":"Aggregation consistency and frequency of Chinese words and characters","year":2006,"lang":"en","type":"article","venue":"Journal of Documentation","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Zipf's law; Consistency (knowledge bases); Syllable; Romanization; Vernacular; Distribution (mathematics); Frequency distribution; Set (abstract data type); Mathematics; Computer science; Econometrics; Natural language processing; Data set; Statistics; Linguistics; Artificial intelligence; Speech recognition","score_opus":0.004694726306288027,"score_gpt":0.2702126740971309,"score_spread":0.26551794779084287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978265849","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98463213,0.00028978277,0.007835521,0.00009279475,0.00002643316,0.00008306446,0.0016201119,0.000100091434,0.0053199953],"genre_scores_gemma":[0.9965222,0.0000588211,0.0019479784,0.000014849472,0.000018200062,0.000072650575,0.0011232534,0.0000149090265,0.00022721251],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.9846631,0.003506273,0.003028985,0.00260495,0.0057525025,0.00044417573],"domain_scores_gemma":[0.93607134,0.032135487,0.013917564,0.009441185,0.008006214,0.00042820306],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008071297,0.0002746798,0.0006870587,0.008721721,0.0009123872,0.0021679045,0.0007415465,0.00030411335,0.0025526287],"category_scores_gemma":[0.062020183,0.00019240574,0.0006233651,0.015455048,0.0015159076,0.0016874068,0.001531154,0.00036498826,0.00039277182],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003321465,0.000054240147,0.8967346,0.0005509155,0.00045576814,0.00031733836,0.0056443987,0.0021130012,0.0030433745,0.0046760472,0.0009789879,0.0850992],"study_design_scores_gemma":[0.000011835368,0.00010546114,0.9824869,0.00007774704,0.00012201606,0.00035208132,0.0020013591,0.006175267,0.0023736248,0.0037141903,0.0025330982,0.00004640788],"about_ca_topic_score_codex":0.0040357998,"about_ca_topic_score_gemma":0.0028335722,"teacher_disagreement_score":0.008721721,"about_ca_system_score_codex":0.0013078881,"about_ca_system_score_gemma":0.0007549166,"threshold_uncertainty_score":0.042685628},"labels":[],"label_agreement":null},{"id":"W1978406346","doi":"10.3758/bf03192769","title":"Anagram software for cognitive research that enables specification of psycholinguistic variables","year":2006,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Anagram; Anagrams; Computer science; Task (project management); Software; Cognition; Natural language processing; Artificial intelligence; Programming language; Psychology","score_opus":0.46341037151236264,"score_gpt":0.6295342610735168,"score_spread":0.1661238895611542,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978406346","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006886437,0.000073399264,0.81842375,0.00012443935,0.00019406542,0.0011115673,0.014426761,0.14951351,0.009246013],"genre_scores_gemma":[0.057148438,0.00015702323,0.8748867,0.00027848152,0.00012574096,0.011301609,0.011934049,0.028422132,0.015745796],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99815387,0.000652772,0.00038540867,0.000390664,0.00031060845,0.00010676769],"domain_scores_gemma":[0.9792772,0.016287876,0.00082818477,0.0019067534,0.0014395387,0.0002604963],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033961218,0.00153074,0.0010296847,0.0021571652,0.0008215264,0.0018433172,0.0012426511,0.0007611942,0.06974609],"category_scores_gemma":[0.017902935,0.001099353,0.0013178092,0.0014961953,0.0005243969,0.0021812548,0.001998598,0.0018324257,0.014597106],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002201992,0.0008484734,0.009289306,0.003580199,0.000497334,0.0007513474,0.003082593,0.0035185418,0.038155086,0.09710042,0.22733061,0.6136441],"study_design_scores_gemma":[0.0012892743,0.0008571522,0.026240803,0.00067648944,0.000858313,0.0020192943,0.00076505955,0.12124303,0.0867445,0.20498553,0.5539744,0.00034606937],"about_ca_topic_score_codex":0.0010777029,"about_ca_topic_score_gemma":0.0016102003,"teacher_disagreement_score":0.06974609,"about_ca_system_score_codex":0.00055008364,"about_ca_system_score_gemma":0.0015897236,"threshold_uncertainty_score":0.23332393},"labels":[],"label_agreement":null},{"id":"W1978683092","doi":"10.1177/016224390202700301","title":"From Thing to Sign and “Natural Object”: Toward a Genetic Phenomenology of Graph Interpretation","year":2002,"lang":"en","type":"article","venue":"Science Technology & Human Values","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières; Lakehead University; University of Victoria","funders":"","keywords":"Interpretation (philosophy); Epistemology; Natural science; Phenomenology (philosophy); Reading (process); Natural (archaeology); Object (grammar); Sign (mathematics); Computer science; Semiotics; Cognitive science; Sociology; Linguistics; Psychology; Artificial intelligence; Mathematics; Philosophy; History","score_opus":0.01471340560740698,"score_gpt":0.2846390009863392,"score_spread":0.26992559537893224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978683092","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5593387,0.0007694038,0.3242385,0.015721906,0.00014788064,0.00026401723,0.0001710809,0.00022708433,0.09912146],"genre_scores_gemma":[0.96872216,0.00018177365,0.028619125,0.00034598293,0.000013620471,0.00006408285,0.000047785474,0.00007910611,0.0019263288],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9964563,0.002697813,0.000058963516,0.00035668167,0.00028905712,0.00014125052],"domain_scores_gemma":[0.9929367,0.005299094,0.0005035005,0.0006227784,0.0003474434,0.00029054505],"candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.004943949,0.0003687608,0.00029912285,0.002047853,0.0028775353,0.006969021,0.0012260228,0.00209269,0.0021664493],"category_scores_gemma":[0.013399682,0.0004609296,0.00047287787,0.0014892118,0.035760656,0.014886659,0.003460974,0.0032996926,0.00027718136],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054199438,0.00004665983,0.0037452865,0.000076431126,0.000008906976,0.00044990866,0.3564538,0.0008586989,0.0028306907,0.6266785,0.0004977694,0.008299088],"study_design_scores_gemma":[0.00002801196,0.000065749344,0.0023719387,0.000088102075,0.000016411945,0.00060768897,0.16264029,0.0061730696,0.0014702955,0.8051632,0.021324238,0.000051024883],"about_ca_topic_score_codex":0.0032322186,"about_ca_topic_score_gemma":0.0022772711,"teacher_disagreement_score":0.99712247,"about_ca_system_score_codex":0.0027685552,"about_ca_system_score_gemma":0.0015326695,"threshold_uncertainty_score":0.026146412},"labels":[],"label_agreement":null},{"id":"W1979229208","doi":"10.5555/2025756.2025769","title":"Multimodal representations, indexing, unexpectedness and proteins","year":2011,"lang":"en","type":"article","venue":"International Conference Industrial, Engineering & Other Applications Applied Intelligent Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Modalities; Computer science; Search engine indexing; Human–computer interaction; Topology (electrical circuits); Artificial intelligence; Computational biology; Biology; Engineering","score_opus":0.07768854023297872,"score_gpt":0.2903790542009867,"score_spread":0.21269051396800798,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1979229208","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7244234,0.005680627,0.24184741,0.0025110506,0.00035471396,0.00010273492,0.003966767,0.001759996,0.019353367],"genre_scores_gemma":[0.96270543,0.00096861075,0.029440176,0.00013137943,0.00016940307,0.000051279927,0.0019370207,0.00008488021,0.0045117945],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99951005,0.0001078936,0.000041642612,0.00014519307,0.00012280598,0.000072492374],"domain_scores_gemma":[0.9981275,0.0007700094,0.00047166794,0.00023289809,0.00028312235,0.00011482964],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082690787,0.00046328496,0.000503497,0.0021782792,0.00068427075,0.0024752321,0.0005471968,0.000896991,0.0052730287],"category_scores_gemma":[0.005146838,0.00018227534,0.00046373383,0.0027281216,0.00071432104,0.0028338092,0.0011040407,0.00057273544,0.00090405735],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030130767,0.00045633086,0.024605054,0.0013519115,0.00029149718,0.001695952,0.0017441069,0.029674504,0.14510363,0.13644116,0.015489835,0.64013296],"study_design_scores_gemma":[0.00013075148,0.0007830386,0.04772719,0.00022420552,0.00035481044,0.003173361,0.0021957615,0.44156146,0.04281552,0.43902746,0.021780362,0.0002261317],"about_ca_topic_score_codex":0.0010730538,"about_ca_topic_score_gemma":0.0011589123,"teacher_disagreement_score":0.0052730287,"about_ca_system_score_codex":0.0006770923,"about_ca_system_score_gemma":0.00040050168,"threshold_uncertainty_score":0.017639995},"labels":[],"label_agreement":null},{"id":"W1986494226","doi":"10.1109/services.2013.62","title":"PALTask Chat: A Personalized Automated Context Aware Web Resources Listing Tool","year":2013,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"National Research Council Canada; Natural Sciences and Engineering Research Council of Canada; University of Victoria","keywords":"Computer science; World Wide Web; Task (project management); The Internet; Context (archaeology); Listing (finance); Domain (mathematical analysis); User profile; Multimedia; Information retrieval; Human–computer interaction","score_opus":0.01204970316159229,"score_gpt":0.2599368146437869,"score_spread":0.24788711148219458,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1986494226","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11552198,0.0016418245,0.51256335,0.0005340889,0.0004328163,0.0012842182,0.0035161166,0.3487898,0.015715748],"genre_scores_gemma":[0.5419017,0.0006569729,0.4139051,0.000625651,0.00031745253,0.0011236508,0.005484505,0.005682816,0.030302027],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993148,0.00023039439,0.000054773063,0.00015849885,0.00017548354,0.0000660426],"domain_scores_gemma":[0.99819726,0.00094644807,0.00013573252,0.00029505568,0.00018538923,0.00024009596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008141168,0.0013888348,0.00083743583,0.0015633918,0.0006607399,0.0009837197,0.0015308311,0.0010374261,0.009347983],"category_scores_gemma":[0.0034469652,0.00042901008,0.00048397918,0.00047217048,0.00026027998,0.0017252836,0.002022904,0.0007194099,0.0037534714],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0043798094,0.0015099957,0.0073211724,0.0019129728,0.0003739155,0.0043502287,0.003640601,0.0047574313,0.14835887,0.004106935,0.10338442,0.71590376],"study_design_scores_gemma":[0.0016949696,0.0032059278,0.038634997,0.0006385975,0.0007841887,0.007529521,0.004062467,0.45652685,0.16438034,0.014696177,0.30704263,0.00080338673],"about_ca_topic_score_codex":0.0013180419,"about_ca_topic_score_gemma":0.002476584,"teacher_disagreement_score":0.009347983,"about_ca_system_score_codex":0.00021742063,"about_ca_system_score_gemma":0.00049537135,"threshold_uncertainty_score":0.031272113},"labels":[],"label_agreement":null},{"id":"W1988834855","doi":"10.1002/meet.1450420170","title":"MARTT: Using induced knowledge base to automatically mark up plant taxonomic descriptions with XML","year":2005,"lang":"en","type":"article","venue":"Proceedings of the American Society for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Markup language; XML; Computer science; RuleML; Domain (mathematical analysis); Knowledge base; XHTML; Natural language processing; Artificial intelligence; Information retrieval; World Wide Web; Mathematics","score_opus":0.02088201374965352,"score_gpt":0.2770650099305437,"score_spread":0.2561829961808902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1988834855","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06385507,0.00024454546,0.87130624,0.00041553023,0.00013334039,0.0003935221,0.00403233,0.05579398,0.0038254056],"genre_scores_gemma":[0.15399057,0.00019564044,0.83262104,0.00016614006,0.000039934348,0.00026095333,0.009682116,0.00056190573,0.0024816953],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992988,0.00020588454,0.000084216175,0.00017149128,0.00021351107,0.000026072239],"domain_scores_gemma":[0.99581134,0.002448686,0.0003934016,0.00067892595,0.00058523833,0.000082458064],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010822368,0.00048342123,0.00039557007,0.0025525594,0.00050071755,0.001264661,0.0014691765,0.00079494744,0.003537841],"category_scores_gemma":[0.0070676757,0.00033445837,0.0006874911,0.0013915356,0.00038917217,0.0025117788,0.0010012143,0.0008702368,0.0013568249],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004170627,0.0004540356,0.0034151506,0.00068448385,0.0001535377,0.00095768983,0.000609522,0.048687585,0.031086843,0.012313887,0.024591265,0.8766289],"study_design_scores_gemma":[0.00009346237,0.0002223921,0.0023334995,0.00014812926,0.00010818358,0.00065008464,0.0003263615,0.84812146,0.09317507,0.014321242,0.040424366,0.000075827425],"about_ca_topic_score_codex":0.0035176708,"about_ca_topic_score_gemma":0.004081715,"teacher_disagreement_score":0.003537841,"about_ca_system_score_codex":0.0006811526,"about_ca_system_score_gemma":0.0010566239,"threshold_uncertainty_score":0.0118352175},"labels":[],"label_agreement":null},{"id":"W1989265129","doi":"10.1108/17440081011090220","title":"Topic‐based web site summarization","year":2010,"lang":"en","type":"article","venue":"International Journal of Web Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Automatic summarization; Computer science; Multi-document summarization; Information retrieval; Cluster analysis; Web page; Set (abstract data type); Web modeling; World Wide Web; Artificial intelligence","score_opus":0.006202646875641511,"score_gpt":0.26099236416118304,"score_spread":0.2547897172855415,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1989265129","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028706318,0.0019782446,0.9517299,0.00045611613,0.00023028928,0.0007823401,0.0024833262,0.010511556,0.0031218743],"genre_scores_gemma":[0.23671427,0.0014913473,0.7407585,0.00018036642,0.0007338832,0.0007426371,0.011006574,0.0009954494,0.007377072],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986381,0.00035340962,0.00015424576,0.00034876188,0.0004243962,0.00008100809],"domain_scores_gemma":[0.99470925,0.001525897,0.00067604426,0.0006376768,0.0022844623,0.00016673961],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017611855,0.0010735568,0.0012561806,0.005275275,0.00074559334,0.002384625,0.0014198093,0.0006914644,0.0027906839],"category_scores_gemma":[0.006972556,0.00035582986,0.00082498195,0.00362141,0.00030381425,0.002404574,0.0010418781,0.0008471564,0.0020674642],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031369916,0.00015858811,0.003612031,0.0009505472,0.00019311723,0.00021131361,0.0008824539,0.013708037,0.037324656,0.0038780144,0.01954065,0.919227],"study_design_scores_gemma":[0.00016074291,0.00080832513,0.018495796,0.00035837753,0.0009108387,0.0013251734,0.0021562162,0.7632228,0.09120248,0.025088387,0.09599441,0.0002763794],"about_ca_topic_score_codex":0.002260032,"about_ca_topic_score_gemma":0.0030160975,"teacher_disagreement_score":0.005275275,"about_ca_system_score_codex":0.00052482716,"about_ca_system_score_gemma":0.0009514572,"threshold_uncertainty_score":0.009335816},"labels":[],"label_agreement":null},{"id":"W1989573439","doi":"10.1109/icassp.2014.6853571","title":"Improving dialogue classification using a topic space representation and a Gaussian classifier based on the decision rule","year":2014,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Artificial intelligence; Classifier (UML); Decision rule; Gaussian; Representation (politics); Space (punctuation); Natural language processing; Machine learning; Pattern recognition (psychology)","score_opus":0.053622232715271494,"score_gpt":0.3238928781536958,"score_spread":0.2702706454384243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1989573439","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04794488,0.0005152545,0.948764,0.00023192802,0.000097502554,0.0000663318,0.00007246328,0.0015275612,0.0007799892],"genre_scores_gemma":[0.5770805,0.00032070375,0.41872418,0.00020424262,0.000269416,0.00018272798,0.00059039344,0.00023731885,0.0023904038],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958948,0.0019067775,0.00022203312,0.0008510851,0.00080384436,0.00032151182],"domain_scores_gemma":[0.9929711,0.0048144734,0.00024770224,0.0004684671,0.0013009676,0.00019712323],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054354616,0.0010866418,0.0018683813,0.0016974824,0.0006528106,0.0020488522,0.0016406446,0.0018417106,0.0012828851],"category_scores_gemma":[0.013034972,0.00035094694,0.0011946174,0.001381299,0.0006818144,0.0026915018,0.0011851573,0.0024446028,0.001584553],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000989325,0.00045424982,0.0039309966,0.00022520863,0.00018794683,0.00014195402,0.00068293087,0.109040834,0.019354168,0.0057251984,0.0039748256,0.8552923],"study_design_scores_gemma":[0.000022301338,0.0000761411,0.00074885937,0.00000993779,0.000031509728,0.000041661802,0.00006424864,0.99287415,0.0034017093,0.0020985093,0.0006140998,0.000016984013],"about_ca_topic_score_codex":0.0043010376,"about_ca_topic_score_gemma":0.0023163103,"teacher_disagreement_score":0.0054354616,"about_ca_system_score_codex":0.0008398624,"about_ca_system_score_gemma":0.0010534779,"threshold_uncertainty_score":0.02874583},"labels":[],"label_agreement":null},{"id":"W1989994111","doi":"10.1145/1502650.1502696","title":"A multimedia interface for facilitating comparisons of opinions","year":2009,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Interface (matter); Visualization; Set (abstract data type); Information visualization; Multimedia; Baseline (sea); User interface; Human–computer interaction; Information retrieval; World Wide Web; Artificial intelligence","score_opus":0.033660982285267906,"score_gpt":0.3613166909410079,"score_spread":0.32765570865573995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1989994111","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13975656,0.0010827279,0.70121485,0.0012105786,0.00046610134,0.003564054,0.013313527,0.10542281,0.03396887],"genre_scores_gemma":[0.2997477,0.00068179873,0.66238713,0.0012138552,0.00053773855,0.0036198231,0.008310921,0.0036226513,0.019878412],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992817,0.0003174314,0.00006890891,0.00012102878,0.00017037577,0.00004068042],"domain_scores_gemma":[0.99134785,0.0068444,0.00024123694,0.0003334277,0.0008693732,0.0003637908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020856457,0.001140589,0.0005104756,0.0018684169,0.00038923876,0.0012752704,0.0009053279,0.00112555,0.051831923],"category_scores_gemma":[0.011520181,0.00022498584,0.0003068144,0.0010024833,0.00021581091,0.0020801804,0.0010372279,0.00043376174,0.0066296314],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002984301,0.00060981564,0.0035065205,0.0014273574,0.00007575752,0.0009017813,0.0034642122,0.00077407056,0.08835686,0.004741119,0.14061683,0.7525413],"study_design_scores_gemma":[0.002842168,0.005349841,0.04710335,0.001263399,0.00057799503,0.0046980805,0.005042879,0.11399557,0.10736687,0.035561834,0.67546606,0.00073208637],"about_ca_topic_score_codex":0.000496727,"about_ca_topic_score_gemma":0.0007753921,"teacher_disagreement_score":0.051831923,"about_ca_system_score_codex":0.000260492,"about_ca_system_score_gemma":0.0003070008,"threshold_uncertainty_score":0.1733951},"labels":[],"label_agreement":null},{"id":"W1991664945","doi":"10.3166/ria.24.97-120","title":"Centering Information Retrieval to the User","year":2010,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Terminology; Domain (mathematical analysis); Humanities; Information retrieval; Artificial intelligence; Linguistics; Philosophy; Mathematics","score_opus":0.031751355571021996,"score_gpt":0.293792729742198,"score_spread":0.262041374171176,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1991664945","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007916407,0.00521627,0.86447537,0.013332062,0.00054553413,0.0002705168,0.00011996316,0.003244326,0.10487958],"genre_scores_gemma":[0.277497,0.009566571,0.5545156,0.00938073,0.0029223673,0.0007040081,0.0009801722,0.0023813709,0.14205216],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99163014,0.0036918633,0.00042088222,0.0014944862,0.002400469,0.00036208893],"domain_scores_gemma":[0.98919135,0.0047386605,0.0003473388,0.0036392133,0.0016509554,0.00043237276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062882863,0.0013257179,0.0019418852,0.003454216,0.0025263173,0.013937423,0.0027442547,0.0030482118,0.026631389],"category_scores_gemma":[0.016720055,0.0008464403,0.0018213816,0.0030791587,0.0046264078,0.019567976,0.009791182,0.004252957,0.014011825],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030128958,0.00027317257,0.0016287335,0.0012634841,0.0002116419,0.00033671642,0.0068062684,0.0024043417,0.013301728,0.38536674,0.026297206,0.56180865],"study_design_scores_gemma":[0.00014764673,0.00039427768,0.001346128,0.00063096016,0.00029353597,0.0009725068,0.0025314773,0.022674602,0.01901345,0.2852276,0.66660005,0.00016770678],"about_ca_topic_score_codex":0.0028567635,"about_ca_topic_score_gemma":0.0028559188,"teacher_disagreement_score":0.026631389,"about_ca_system_score_codex":0.0024935945,"about_ca_system_score_gemma":0.002785214,"threshold_uncertainty_score":0.089090884},"labels":[],"label_agreement":null},{"id":"W1992760963","doi":"10.1109/bibm.2013.6732593","title":"Innovative navigation of health discussion forums based on relationship extraction and medical ontologies","year":2013,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Ontology; Computer science; Unified Medical Language System; Interface (matter); Natural language; Information extraction; World Wide Web; Information retrieval; Open Biomedical Ontologies; Data science; Natural language processing; Upper ontology; Semantic Web; Ontology alignment","score_opus":0.02783934920412459,"score_gpt":0.3511083292126822,"score_spread":0.32326898000855764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1992760963","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034032524,0.0004792335,0.9444025,0.0011819048,0.00021841098,0.0006599977,0.0016055589,0.010424213,0.006995643],"genre_scores_gemma":[0.09582578,0.00023709428,0.8967481,0.00018072288,0.000055156386,0.0003528033,0.0021732005,0.0005004237,0.0039266893],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99673945,0.0014417173,0.00042344513,0.0005941138,0.0006534597,0.00014772113],"domain_scores_gemma":[0.9925356,0.0046194578,0.0006540472,0.0006518148,0.0011940479,0.0003450026],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035335904,0.0009511871,0.00072189566,0.0070342547,0.0018653647,0.0029887352,0.0011897418,0.0013962786,0.006452228],"category_scores_gemma":[0.011510981,0.0005070338,0.0011823035,0.003964064,0.00067914056,0.007014306,0.0035706686,0.0008604693,0.0016347201],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072714157,0.00065993326,0.009209739,0.0016755337,0.00013946618,0.0011089864,0.012440525,0.0054103765,0.040752944,0.08278431,0.029559761,0.8155313],"study_design_scores_gemma":[0.00033019774,0.0004232765,0.010516781,0.0008687406,0.00033476742,0.002033747,0.00920302,0.36085376,0.06972026,0.18770263,0.3576056,0.00040714646],"about_ca_topic_score_codex":0.002716663,"about_ca_topic_score_gemma":0.0039347457,"teacher_disagreement_score":0.0070342547,"about_ca_system_score_codex":0.00077909435,"about_ca_system_score_gemma":0.001919137,"threshold_uncertainty_score":0.021584868},"labels":[],"label_agreement":null},{"id":"W1996606535","doi":"10.1037/0278-7393.32.6.1244","title":"Linking associative and serial list memory: Pairs versus triples.","year":2006,"lang":"en","type":"article","venue":"Journal of Experimental Psychology Learning Memory and Cognition","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Baycrest Hospital","funders":"Canadian Institutes of Health Research","keywords":"Dissociation (chemistry); Computer science; Associative property; Content-addressable memory; Isolation (microbiology); Arithmetic; Artificial intelligence; Mathematics; Biology; Pure mathematics; Chemistry; Artificial neural network","score_opus":0.022152592808653507,"score_gpt":0.3259258794838167,"score_spread":0.3037732866751632,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1996606535","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90310335,0.0014575076,0.07530591,0.0006488096,0.00025119164,0.00023055282,0.000290512,0.00021702926,0.018495256],"genre_scores_gemma":[0.9863369,0.00037391752,0.011365286,0.00026631015,0.00006320277,0.00014427824,0.000203689,0.000029799718,0.0012165152],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99848026,0.00046013732,0.00008594869,0.0004003803,0.0005030045,0.00007030151],"domain_scores_gemma":[0.98534346,0.0077379732,0.0025738163,0.002894924,0.0008859996,0.0005638468],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024767607,0.000517802,0.000751374,0.0010946023,0.00047618555,0.001870782,0.0010317354,0.00094253605,0.004440105],"category_scores_gemma":[0.023863507,0.0004890593,0.00052127993,0.00094673014,0.0018842474,0.007296265,0.0023219527,0.0017668658,0.00061585684],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010152521,0.0016089899,0.104757756,0.0020199392,0.0010392736,0.001020527,0.0069903163,0.0062311413,0.17856444,0.15880977,0.0033199205,0.52548546],"study_design_scores_gemma":[0.0005423844,0.0052321153,0.17852412,0.00030680277,0.0011371159,0.0039024556,0.0029594281,0.07373127,0.1572044,0.56637275,0.009672446,0.00041467347],"about_ca_topic_score_codex":0.00055085594,"about_ca_topic_score_gemma":0.0005572854,"teacher_disagreement_score":0.004440105,"about_ca_system_score_codex":0.0005702369,"about_ca_system_score_gemma":0.0005282832,"threshold_uncertainty_score":0.014853597},"labels":[],"label_agreement":null},{"id":"W2001124886","doi":"10.1300/j104v28n04_05","title":"The Essential Elements of Faceted Thesauri","year":2000,"lang":"en","type":"article","venue":"Cataloging & Classification Quarterly","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Information retrieval; Facet (psychology); Construct (python library); Search engine indexing; Selection (genetic algorithm); Homogeneous; Artificial intelligence; Mathematics","score_opus":0.01282327803168236,"score_gpt":0.2720939766783442,"score_spread":0.2592706986466618,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2001124886","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13478512,0.00339937,0.7927847,0.002269325,0.00023623998,0.002462513,0.0015489917,0.0021444198,0.060369402],"genre_scores_gemma":[0.52497447,0.0013809913,0.46621665,0.00023597624,0.00012247635,0.0010756331,0.0012613354,0.00038199493,0.0043505225],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9773351,0.012276624,0.0032995956,0.00092203286,0.005759335,0.00040726463],"domain_scores_gemma":[0.8981678,0.053961366,0.0064372965,0.01736076,0.023051603,0.0010212209],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014396152,0.00046930116,0.00064981927,0.007952268,0.0027245954,0.010790169,0.0009804926,0.00060597464,0.0028229982],"category_scores_gemma":[0.091696516,0.00053165556,0.0009818835,0.00950146,0.0039577037,0.007187844,0.0026529017,0.0013222774,0.0010274599],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034887262,0.00012917387,0.016721122,0.0027213988,0.00017181571,0.00031744683,0.046031784,0.0039365203,0.018249303,0.18380807,0.0066809626,0.7208835],"study_design_scores_gemma":[0.00008976355,0.00089759234,0.06162245,0.004823893,0.00065934705,0.00197209,0.03798302,0.06724903,0.028889766,0.4752663,0.32008132,0.00046536734],"about_ca_topic_score_codex":0.006586083,"about_ca_topic_score_gemma":0.004852015,"teacher_disagreement_score":0.014396152,"about_ca_system_score_codex":0.0029638684,"about_ca_system_score_gemma":0.004659849,"threshold_uncertainty_score":0.07613498},"labels":[],"label_agreement":null},{"id":"W2002945463","doi":"10.1145/2682571.2797083","title":"Enhancing Exploration with a Faceted Browser through Summarization","year":2015,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"University of Waterloo; Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Automatic summarization; Computer science; Information retrieval; Set (abstract data type); Centroid; Multi-document summarization; World Wide Web; Thesaurus; Natural language processing; Artificial intelligence; Programming language","score_opus":0.04618328961949001,"score_gpt":0.2866778724386565,"score_spread":0.2404945828191665,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002945463","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06512715,0.0008733519,0.8433703,0.00028857277,0.00005680899,0.00024370465,0.0030089621,0.08194729,0.005083833],"genre_scores_gemma":[0.21889755,0.0006073693,0.7585275,0.00015869262,0.00007873338,0.00031123238,0.010004053,0.0033137277,0.008101023],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992893,0.00023938253,0.00007876822,0.00012958168,0.00022090314,0.000042112162],"domain_scores_gemma":[0.9953366,0.0025298684,0.00017810451,0.00067063933,0.0011121781,0.00017264753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015594228,0.00081680185,0.00084667746,0.00232977,0.0004702523,0.0020913759,0.00079679745,0.0006540745,0.0054272725],"category_scores_gemma":[0.005842298,0.0003154354,0.0004553645,0.0023070616,0.00027156417,0.00233757,0.0019111672,0.00061557395,0.0023872536],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012198926,0.00023069758,0.0053694504,0.0010107842,0.0002000539,0.0011930877,0.0054691695,0.0096178,0.14230274,0.0061429483,0.041854292,0.7853891],"study_design_scores_gemma":[0.00044283894,0.0006863155,0.011157049,0.0003771808,0.0004166368,0.0028647198,0.0020464952,0.60690445,0.14422745,0.025858594,0.20452559,0.00049265765],"about_ca_topic_score_codex":0.0030666105,"about_ca_topic_score_gemma":0.006267152,"teacher_disagreement_score":0.0054272725,"about_ca_system_score_codex":0.00027708043,"about_ca_system_score_gemma":0.0007107472,"threshold_uncertainty_score":0.018156052},"labels":[],"label_agreement":null},{"id":"W2004269233","doi":"10.5210/fm.v18i5.4529","title":"Navigating an imagined Middle&amp;ndash;earth: Finding and analyzing text&amp;ndash;based and film&amp;ndash;based mental images of Middle&amp;ndash;earth through TheOneRing.net online fan community","year":2013,"lang":"en","type":"article","venue":"First Monday","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Point (geometry); Adaptation (eye); Mental image; Dissemination; Social media; World Wide Web; Computer science; Sociology; Psychology; Cognition; Media studies; Mathematics","score_opus":0.054980428905124085,"score_gpt":0.3176600709940333,"score_spread":0.26267964208890926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2004269233","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9657908,0.00022178459,0.020081874,0.00061493274,0.00003059222,0.00007454985,0.0003170936,0.00026625625,0.012602041],"genre_scores_gemma":[0.962153,0.0002087845,0.031112581,0.000069054666,0.000015476155,0.00003422224,0.0004110707,0.00009393323,0.005901837],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9997613,0.000094129224,0.000006802084,0.00006588146,0.000046628862,0.000025279365],"domain_scores_gemma":[0.99902165,0.00054652034,0.00013214591,0.0000796052,0.00009974544,0.000120194745],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006091798,0.00029254137,0.00011743679,0.0019202132,0.001174476,0.0030299213,0.0003724392,0.00041271414,0.003193181],"category_scores_gemma":[0.0025258486,0.00012591959,0.00017381209,0.00094189326,0.0012921263,0.0038697754,0.001130563,0.00039393274,0.0005706737],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051258487,0.0003782921,0.094899096,0.0004501518,0.00006326054,0.001673372,0.42828158,0.0011018034,0.032869615,0.013921753,0.014563862,0.41128471],"study_design_scores_gemma":[0.000038559276,0.0002511706,0.17437348,0.00032664902,0.00007742957,0.0013315846,0.6251714,0.046343643,0.021120008,0.024715543,0.10607863,0.00017193235],"about_ca_topic_score_codex":0.0045688045,"about_ca_topic_score_gemma":0.014356755,"teacher_disagreement_score":0.0045688045,"about_ca_system_score_codex":0.0005680805,"about_ca_system_score_gemma":0.00041877574,"threshold_uncertainty_score":0.010682225},"labels":[],"label_agreement":null},{"id":"W2005025726","doi":"10.5539/ells.v1n2p129","title":"An Innovative Way of Finding Best or Least Matching Pairs and Groups","year":2011,"lang":"en","type":"article","venue":"English Language and Literature Studies","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Matching (statistics); Rank (graph theory); Pairing; Mathematics; Group (periodic table); Variable (mathematics); Computer science; Combinatorics; Statistics; Physics; Mathematical analysis","score_opus":0.0213697606517436,"score_gpt":0.29725035190274685,"score_spread":0.2758805912510032,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2005025726","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0060279532,0.000053680746,0.98760486,0.00027715735,0.00017903927,0.001535233,0.00044123764,0.0010139832,0.0028668228],"genre_scores_gemma":[0.01266346,0.000019445639,0.9846517,0.00007852662,0.000029361701,0.0013008548,0.00023431693,0.00016927959,0.00085319625],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9541682,0.02732382,0.0044494746,0.006807575,0.006318375,0.0009325769],"domain_scores_gemma":[0.94429076,0.031323522,0.0033662478,0.011100506,0.009204785,0.00071418175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02907398,0.002005632,0.002530944,0.009822488,0.004236517,0.0047395756,0.0038727464,0.0024853153,0.025224544],"category_scores_gemma":[0.1251557,0.0015120951,0.003220419,0.007812884,0.00340194,0.0061072567,0.0055682035,0.0033145705,0.006314511],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014538171,0.00059512065,0.005635215,0.0018631758,0.00066457625,0.00040242504,0.0115871,0.00333172,0.015740462,0.15019746,0.024024747,0.7845042],"study_design_scores_gemma":[0.0015111578,0.0027091333,0.011883452,0.0011145257,0.0010665507,0.0029532684,0.014698938,0.07318706,0.04637897,0.5972832,0.24614397,0.0010697427],"about_ca_topic_score_codex":0.00085199455,"about_ca_topic_score_gemma":0.0013310702,"teacher_disagreement_score":0.02907398,"about_ca_system_score_codex":0.0012159866,"about_ca_system_score_gemma":0.003922885,"threshold_uncertainty_score":0.15375978},"labels":[],"label_agreement":null},{"id":"W2007176642","doi":"10.1002/meet.14504701334","title":"Newsblog relevance: Applying relevance criteria to news‐related blogs","year":2010,"lang":"en","type":"article","venue":"Proceedings of the American Society for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Relevance (law); Ranking (information retrieval); Quality (philosophy); Psychology; Computer science; Information retrieval; Political science; Epistemology","score_opus":0.007204722471592029,"score_gpt":0.2823218099709089,"score_spread":0.27511708749931685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2007176642","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89648795,0.0014708526,0.08293903,0.00059342023,0.00018913709,0.0025227924,0.0017876179,0.0015262727,0.012482959],"genre_scores_gemma":[0.94222444,0.00014020393,0.055603035,0.0000507848,0.00007152468,0.000501491,0.00062289333,0.00008805189,0.00069764676],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9838344,0.0068053547,0.0020844492,0.001269908,0.005459401,0.000546493],"domain_scores_gemma":[0.8199926,0.14570238,0.009241783,0.0036382712,0.018633943,0.0027911041],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014129425,0.00062487955,0.001083015,0.015516586,0.0018819369,0.0047009727,0.00086932356,0.0014246643,0.0021795356],"category_scores_gemma":[0.14644982,0.0003309358,0.00091764785,0.008988827,0.0011451456,0.0039644027,0.002177791,0.001383373,0.0005550936],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004729864,0.0021085006,0.39943454,0.003640434,0.0006859672,0.00057071605,0.010375755,0.0083490005,0.017139785,0.006081125,0.011567371,0.53531694],"study_design_scores_gemma":[0.0008027017,0.00421401,0.6446155,0.0010192121,0.00096177246,0.0012448343,0.015369512,0.25924265,0.025969937,0.024570262,0.021274079,0.0007155197],"about_ca_topic_score_codex":0.0036153416,"about_ca_topic_score_gemma":0.0065650363,"teacher_disagreement_score":0.015516586,"about_ca_system_score_codex":0.0015516756,"about_ca_system_score_gemma":0.0015841593,"threshold_uncertainty_score":0.074724436},"labels":[],"label_agreement":null},{"id":"W2010118061","doi":"10.3758/s13420-013-0112-z","title":"An elemental model of retrospective revaluation without within-compound associations","year":2013,"lang":"en","type":"article","venue":"Learning & Behavior","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Canadian Institutes of Health Research","keywords":"Psychology; Stimulus (psychology); Striatum; Neuroscience; Chemistry; Developmental psychology; Cognitive psychology","score_opus":0.02592354657919084,"score_gpt":0.3318573211196088,"score_spread":0.305933774540418,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2010118061","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15759204,0.0002331113,0.6969345,0.0019832477,0.00022311111,0.00014476335,0.0010192376,0.0017590292,0.14011088],"genre_scores_gemma":[0.89141905,0.00012902265,0.06023274,0.0002969708,0.00010586931,0.00012224172,0.00040228225,0.00045183444,0.04684003],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99893826,0.00025413447,0.00005604475,0.00035730487,0.00026986815,0.00012438482],"domain_scores_gemma":[0.9934064,0.0027929589,0.00043902296,0.0022262726,0.00082580064,0.00030958664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019615141,0.00042788935,0.0006404427,0.000892139,0.0006988232,0.002909744,0.0028573438,0.0017211476,0.026399022],"category_scores_gemma":[0.011583224,0.0007364152,0.0012479834,0.0008127762,0.0017991825,0.009605391,0.0014606488,0.0019242311,0.004097788],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015213844,0.00008347564,0.0014124828,0.00005259895,0.00004324873,0.00024278312,0.0006309422,0.011388287,0.0022647697,0.9544191,0.0018940887,0.027415985],"study_design_scores_gemma":[0.00003381572,0.00005561553,0.0010550646,0.000016217025,0.00004321971,0.00034496837,0.00013099142,0.12582953,0.0011386763,0.86792195,0.0033990738,0.00003089737],"about_ca_topic_score_codex":0.0018368318,"about_ca_topic_score_gemma":0.0013295726,"teacher_disagreement_score":0.026399022,"about_ca_system_score_codex":0.0007890975,"about_ca_system_score_gemma":0.0011224989,"threshold_uncertainty_score":0.08831352},"labels":[],"label_agreement":null},{"id":"W2013295704","doi":"10.1016/s0278-2626(02)00506-7","title":"The referencing of internet web sites in medical and scientific publications","year":2002,"lang":"en","type":"article","venue":"Brain and Cognition","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Search engine; Information retrieval; Set (abstract data type); Ranking (information retrieval); Computer science; The Internet; Similarity (geometry); Measure (data warehouse); Web search engine; Webometrics; Web search query; Spamdexing; Data mining; World Wide Web; Artificial intelligence","score_opus":0.031863293453405044,"score_gpt":0.2794154786655812,"score_spread":0.24755218521217615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2013295704","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9096111,0.012006419,0.01660436,0.0031297905,0.0010902248,0.00014400572,0.0063890577,0.0005422201,0.05048281],"genre_scores_gemma":[0.97525275,0.0027255279,0.012937574,0.00028658853,0.00076574733,0.00007355235,0.0034369708,0.00034165164,0.004179659],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9933795,0.0028333254,0.00095513,0.0006295834,0.0019392304,0.0002631627],"domain_scores_gemma":[0.8703725,0.085142896,0.019693429,0.006426261,0.016194012,0.0021709634],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0056722257,0.0002858358,0.00041904807,0.025608806,0.0015034834,0.005928562,0.00095552247,0.0014737436,0.0049794572],"category_scores_gemma":[0.08017351,0.00031562612,0.000427249,0.026655741,0.0014227789,0.0065945494,0.0022752623,0.0009443941,0.0016909409],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017378273,0.00043100247,0.38383397,0.0033606049,0.00065717584,0.0041939695,0.040810857,0.0032355194,0.029660191,0.09908689,0.032867193,0.40012482],"study_design_scores_gemma":[0.00021373703,0.00049488037,0.5804856,0.0028146494,0.0012279863,0.009366064,0.029551353,0.029809507,0.035437,0.11893745,0.19135879,0.00030302265],"about_ca_topic_score_codex":0.0017748803,"about_ca_topic_score_gemma":0.0018786087,"teacher_disagreement_score":0.9943278,"about_ca_system_score_codex":0.001125169,"about_ca_system_score_gemma":0.0009770367,"threshold_uncertainty_score":0.029998004},"labels":[],"label_agreement":null},{"id":"W2015225850","doi":"10.1108/00220411311295315","title":"Nodes and arcs: concept map, semiotics, and knowledge organization","year":2013,"lang":"en","type":"article","venue":"Journal of Documentation","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Semiotics; Terminology; Meaning (existential); Knowledge organization; Domain (mathematical analysis); Perspective (graphical); Computer science; Knowledge management; Epistemology; Sociology; Linguistics; Data science; Artificial intelligence; Mathematics","score_opus":0.0049372449305333365,"score_gpt":0.2683386110452432,"score_spread":0.2634013661147099,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2015225850","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35442948,0.010463434,0.3941444,0.011390237,0.00043706567,0.0005686381,0.00039177644,0.0004999062,0.22767504],"genre_scores_gemma":[0.930709,0.0015832721,0.062496696,0.00013890075,0.000044327866,0.00030409216,0.000097390206,0.00005039972,0.0045758937],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9950198,0.0036850276,0.00018549904,0.00034703972,0.0006148109,0.00014784746],"domain_scores_gemma":[0.98561954,0.010921899,0.0013819371,0.0006153216,0.0009960148,0.00046521323],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045694136,0.0006193317,0.00028745885,0.0059311534,0.0023166332,0.0062198127,0.00077715894,0.0010039619,0.003552663],"category_scores_gemma":[0.013608792,0.00026063135,0.00033905296,0.006374313,0.018724445,0.014174138,0.0030094048,0.0011246835,0.00037012188],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008977463,0.000023245284,0.0033962987,0.00036733097,0.000014219453,0.00027116647,0.062029816,0.0014562844,0.00040223263,0.8852184,0.0012875275,0.04544372],"study_design_scores_gemma":[0.000019384424,0.00004875382,0.0026487196,0.00043453043,0.000021413021,0.0005860192,0.061796438,0.007751335,0.0007455651,0.87377346,0.052133374,0.000041044343],"about_ca_topic_score_codex":0.0033398613,"about_ca_topic_score_gemma":0.0020383028,"teacher_disagreement_score":0.0062198127,"about_ca_system_score_codex":0.0035049056,"about_ca_system_score_gemma":0.0028825253,"threshold_uncertainty_score":0.025429964},"labels":[],"label_agreement":null},{"id":"W2017155419","doi":"10.1016/j.procs.2011.09.050","title":"Quantum Theory-Inspired Search","year":2011,"lang":"en","type":"article","venue":"Procedia Computer Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Engineering and Physical Sciences Research Council; European Commission","keywords":"Computer science; Quantum; Theoretical computer science; Quantum mechanics; Physics","score_opus":0.035995282752279215,"score_gpt":0.28006056234070886,"score_spread":0.24406527958842966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2017155419","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10944381,0.007257853,0.76360923,0.005836222,0.00063637213,0.00024629582,0.00037628366,0.00036615573,0.112227686],"genre_scores_gemma":[0.89992756,0.0024141287,0.08279989,0.000602512,0.00030047284,0.00027678796,0.00018924451,0.00009311881,0.013396392],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938536,0.00025613458,0.000027854696,0.00006793606,0.00019674003,0.00006600884],"domain_scores_gemma":[0.99883026,0.0007588954,0.00007638499,0.00012767581,0.00014342048,0.00006330063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008872083,0.0003128099,0.0009470825,0.0009213757,0.00074894703,0.0015591059,0.0011122598,0.0012294023,0.004528724],"category_scores_gemma":[0.003260529,0.00021275869,0.0007031109,0.0010571707,0.0019569267,0.0024695166,0.0011481822,0.0009554178,0.00041189895],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018103987,0.000016754248,0.00018910323,0.00009349378,0.000024570287,0.000036301757,0.0000838112,0.033916373,0.0005535735,0.95495135,0.0014177308,0.008698836],"study_design_scores_gemma":[0.000026046904,0.000018697812,0.00018125228,0.000013914367,0.000009373559,0.00002932246,0.000029348845,0.33178973,0.00016923774,0.66518503,0.0025336032,0.000014349664],"about_ca_topic_score_codex":0.002519795,"about_ca_topic_score_gemma":0.0015373161,"teacher_disagreement_score":0.004528724,"about_ca_system_score_codex":0.0016965647,"about_ca_system_score_gemma":0.0010349562,"threshold_uncertainty_score":0.01515013},"labels":[],"label_agreement":null},{"id":"W2018222148","doi":"10.1145/1860559.1860615","title":"Structure-aware topic clustering in social media","year":2010,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Conversation; Meaning (existential); Social media; Variety (cybernetics); Set (abstract data type); Cluster analysis; Software; Data science; World Wide Web; Scale (ratio); Artificial intelligence; Sociology; Communication; Psychology","score_opus":0.011197016894605682,"score_gpt":0.27793588576744316,"score_spread":0.2667388688728375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2018222148","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0680647,0.001818362,0.9226403,0.0005891085,0.00014691675,0.00027085852,0.0008813172,0.0036790217,0.0019093452],"genre_scores_gemma":[0.48684177,0.0010305196,0.50203246,0.00015617805,0.00056059245,0.00045253552,0.00393848,0.0006435802,0.004343887],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99617296,0.0013722813,0.00025049632,0.00086764566,0.0009790371,0.00035766396],"domain_scores_gemma":[0.9919207,0.004786947,0.00078514346,0.0009923856,0.0011513665,0.0003634892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041384757,0.0010983422,0.0018505343,0.011047054,0.0022262386,0.0029250917,0.0023215208,0.0017131681,0.0012376551],"category_scores_gemma":[0.012621847,0.0009755131,0.0015571178,0.007677546,0.0010935138,0.0032752436,0.0023132314,0.0011582727,0.0012296109],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009686461,0.0004973378,0.018166194,0.00068544794,0.0005477275,0.00040809708,0.002654914,0.24144822,0.016026022,0.022212131,0.018465983,0.6779193],"study_design_scores_gemma":[0.000050065843,0.00004741278,0.0034156481,0.000028989707,0.000071149494,0.00013459864,0.00043423704,0.96032745,0.004023149,0.02724797,0.0041713123,0.000048061967],"about_ca_topic_score_codex":0.011717932,"about_ca_topic_score_gemma":0.015906421,"teacher_disagreement_score":0.011717932,"about_ca_system_score_codex":0.001374345,"about_ca_system_score_gemma":0.0018793044,"threshold_uncertainty_score":0.023299456},"labels":[],"label_agreement":null},{"id":"W2020664340","doi":"10.1353/lib.2012.0022","title":"Capitalizing on Information Organization and Information Visualization for a New-Generation Catalogue","year":2012,"lang":"en","type":"article","venue":"Library trends","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Centre for Interdisciplinary Research in Music Media and Technology; McGill University; University of Wisconsin-Milwaukee","keywords":"Computer science; Subject (documents); World Wide Web; Information retrieval; USable; Subject access; Digital library; Library catalog; Leverage (statistics); Controlled vocabulary; Vocabulary; Artificial intelligence","score_opus":0.012152060795173256,"score_gpt":0.24596519278730852,"score_spread":0.23381313199213527,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2020664340","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14308389,0.0010881296,0.65544754,0.006960155,0.00021565658,0.0005074819,0.00039100798,0.009222185,0.18308392],"genre_scores_gemma":[0.3418693,0.000669546,0.6367513,0.00047415053,0.000106585016,0.00018789391,0.00063312316,0.0010861236,0.018222023],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9983026,0.00059407356,0.00012576448,0.00021013798,0.00065577094,0.00011167605],"domain_scores_gemma":[0.9900475,0.0029734536,0.0006587098,0.0047228443,0.0010541035,0.0005434312],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0033569199,0.00039859762,0.00035718037,0.003093463,0.0012151338,0.010348483,0.0015232401,0.0008468162,0.007836342],"category_scores_gemma":[0.014124147,0.00062657235,0.00053841394,0.0031964201,0.0023030478,0.01957729,0.0049251667,0.0014442881,0.002493074],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014067968,0.00027171677,0.0071294014,0.0005718196,0.000031576332,0.0005455883,0.020701166,0.0025998421,0.019878695,0.40335125,0.013728681,0.53104955],"study_design_scores_gemma":[0.000082989754,0.0004588542,0.012282187,0.00061452965,0.000120218305,0.0020634413,0.008059308,0.029519727,0.016515192,0.15705569,0.7730446,0.00018326995],"about_ca_topic_score_codex":0.0033412296,"about_ca_topic_score_gemma":0.007648139,"teacher_disagreement_score":0.9896515,"about_ca_system_score_codex":0.0017047337,"about_ca_system_score_gemma":0.0021431728,"threshold_uncertainty_score":0.026215196},"labels":[],"label_agreement":null},{"id":"W2021285674","doi":"10.1109/icalt.2013.158","title":"Knowledge Representation for Context and Sentiment Analysis","year":2013,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University","funders":"Alberta Innovates - Technology Futures","keywords":"Computer science; Representation (politics); Context (archaeology); Knowledge acquisition; Knowledge representation and reasoning; Artificial intelligence; Knowledge extraction; Information extraction; Natural language; Sentiment analysis; Natural language processing; Mechanism (biology); Human–computer interaction","score_opus":0.02252660654375636,"score_gpt":0.3323195221343422,"score_spread":0.3097929155905858,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021285674","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004037731,0.0009079114,0.98618823,0.0006891761,0.000134222,0.00019668114,0.0010252205,0.0014923316,0.0053283744],"genre_scores_gemma":[0.18034163,0.0011682513,0.80875546,0.0003663264,0.0002416331,0.0005066719,0.004673201,0.00019966971,0.0037471473],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982566,0.00063575973,0.00018688412,0.00038731712,0.0004045812,0.00012884686],"domain_scores_gemma":[0.99857223,0.00058369833,0.0001713375,0.00030434504,0.0003131304,0.000055287193],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017998035,0.00078837894,0.0007355216,0.0035249607,0.0010748616,0.0029855699,0.0015632311,0.0010237269,0.0067118797],"category_scores_gemma":[0.00627255,0.00031044145,0.0015624274,0.002538613,0.00087237294,0.0047152643,0.0017209444,0.0015066025,0.0020748905],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000136853,0.00016217433,0.0010127766,0.0006451302,0.0001566048,0.00039230497,0.00087054254,0.01912254,0.009798785,0.2300371,0.021988934,0.7156762],"study_design_scores_gemma":[0.000037792022,0.00006536031,0.0013909002,0.00032918085,0.00013377445,0.00032927928,0.0007114834,0.29416406,0.0102199735,0.58370334,0.10882803,0.00008690453],"about_ca_topic_score_codex":0.004124901,"about_ca_topic_score_gemma":0.0043309038,"teacher_disagreement_score":0.0067118797,"about_ca_system_score_codex":0.001278777,"about_ca_system_score_gemma":0.0014383771,"threshold_uncertainty_score":0.022453427},"labels":[],"label_agreement":null},{"id":"W2023370053","doi":"10.1109/ds-rt.2010.28","title":"Location Aware Question Answering Based Product Searching in Mobile Handheld Devices","year":2010,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Ask price; Conversation; Mobile device; Markup language; Question answering; Global Positioning System; Information retrieval; Product (mathematics); World Wide Web; Human–computer interaction; Word (group theory); Telecommunications; XML","score_opus":0.009268044008883362,"score_gpt":0.30447045181436044,"score_spread":0.2952024078054771,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2023370053","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27413154,0.005072023,0.6990733,0.00086114137,0.00012999294,0.00060729607,0.00076703227,0.009267276,0.0100903325],"genre_scores_gemma":[0.71995485,0.00090950285,0.27031904,0.00043266133,0.00009940397,0.00020790363,0.00084215595,0.00023148512,0.007003001],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988524,0.00042683014,0.00008131329,0.00028715836,0.00025887505,0.00009348852],"domain_scores_gemma":[0.9979037,0.0014240405,0.0001630779,0.00020360589,0.00024341492,0.0000621658],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008070562,0.0005100612,0.0007824987,0.00072493695,0.00065147085,0.0013397378,0.0011119922,0.0013052294,0.0029109379],"category_scores_gemma":[0.0029186863,0.00030345444,0.00054834096,0.00078927755,0.00035281596,0.0027567071,0.00075880124,0.0005465132,0.0015378973],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002272999,0.00052581355,0.0056774626,0.0015756432,0.00020707835,0.0028823214,0.006367689,0.010380142,0.26330984,0.008547133,0.01154373,0.6867101],"study_design_scores_gemma":[0.00034124855,0.0017556682,0.021727955,0.00031592205,0.00056249654,0.004495818,0.0045670765,0.60600466,0.23766531,0.026332663,0.0959213,0.00030985795],"about_ca_topic_score_codex":0.00232933,"about_ca_topic_score_gemma":0.0025084205,"teacher_disagreement_score":0.0029109379,"about_ca_system_score_codex":0.0003565292,"about_ca_system_score_gemma":0.00027306585,"threshold_uncertainty_score":0.009738088},"labels":[],"label_agreement":null},{"id":"W2023684799","doi":"10.3758/bf03196096","title":"Estimating the frequency of events from unnatural categories","year":2003,"lang":"en","type":"article","venue":"Memory & Cognition","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Spinach; Property (philosophy); Event (particle physics); Cognitive psychology; Social psychology; Natural language processing; Computer science; Chemistry; Astrophysics; Epistemology","score_opus":0.01693046206502992,"score_gpt":0.272789210497638,"score_spread":0.2558587484326081,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2023684799","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7371839,0.0015001184,0.25257444,0.00027034778,0.00023281989,0.00012961739,0.0028610541,0.0019712914,0.0032764727],"genre_scores_gemma":[0.9253908,0.00046532828,0.06908152,0.000045013738,0.00016382389,0.00007426963,0.003573123,0.00013036179,0.0010757578],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999084,0.0002183432,0.000089119196,0.00030774708,0.00021550182,0.000085320506],"domain_scores_gemma":[0.9898387,0.007670454,0.0006986106,0.0006007561,0.0009290571,0.0002622412],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013189348,0.0006377535,0.00061025,0.0048648217,0.00042452934,0.001442013,0.0007584474,0.0011644723,0.0021386996],"category_scores_gemma":[0.011317889,0.0002479867,0.0005495292,0.0022379158,0.00038178932,0.0024193977,0.0007409917,0.0010015764,0.0011665601],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036949469,0.00036782416,0.1357404,0.00064734055,0.00033846145,0.00074575405,0.0010891459,0.012735301,0.08101276,0.00558609,0.005173675,0.75286835],"study_design_scores_gemma":[0.000118378746,0.0006510245,0.19108048,0.000118559736,0.00038553332,0.0035196352,0.0013053002,0.73207134,0.040156048,0.022628905,0.007783169,0.00018157095],"about_ca_topic_score_codex":0.0017836036,"about_ca_topic_score_gemma":0.002070676,"teacher_disagreement_score":0.0048648217,"about_ca_system_score_codex":0.00038006314,"about_ca_system_score_gemma":0.00037815433,"threshold_uncertainty_score":0.007154703},"labels":[],"label_agreement":null},{"id":"W2029491700","doi":"10.2307/3315991","title":"Authors' addendum to: “A generalized‐moments specification test for the logistic link”","year":2000,"lang":"en","type":"article","venue":"Canadian Journal of Statistics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Addendum; Link (geometry); Test (biology); Logistic regression; Econometrics; Mathematics; Computer science; Statistics; Political science; Combinatorics; Law; Geology","score_opus":0.05017725175222088,"score_gpt":0.29923536935334927,"score_spread":0.2490581176011284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2029491700","genre_codex":"editorial","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04601445,0.009540889,0.16029488,0.19033037,0.41848528,0.0016424785,0.07719491,0.008000257,0.08849649],"genre_scores_gemma":[0.41588163,0.008348335,0.10635517,0.03702999,0.101784505,0.0026251005,0.08857001,0.004011609,0.23539355],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99446255,0.0018218792,0.0005046454,0.000680159,0.0022865285,0.0002441768],"domain_scores_gemma":[0.8989613,0.0531281,0.0046589524,0.0086979205,0.03278833,0.0017653886],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007630397,0.0009680078,0.0013471968,0.004643045,0.0012315542,0.002012105,0.0036621145,0.0016885671,0.11239388],"category_scores_gemma":[0.123758286,0.000549601,0.0018914883,0.0045201336,0.0010984458,0.0018566031,0.002042654,0.002458283,0.03622927],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018482616,0.00011340979,0.0055956705,0.00024856496,0.00011138087,0.00029469217,0.00008789744,0.0013378075,0.00025733383,0.0047942484,0.9356841,0.051290113],"study_design_scores_gemma":[0.0011060016,0.00088401174,0.05733258,0.0010804671,0.0011022062,0.0019252741,0.0011076266,0.033504646,0.013404194,0.075553045,0.81225926,0.0007407269],"about_ca_topic_score_codex":0.00602441,"about_ca_topic_score_gemma":0.009080505,"teacher_disagreement_score":0.11239388,"about_ca_system_score_codex":0.0010322173,"about_ca_system_score_gemma":0.0029639858,"threshold_uncertainty_score":0.37599498},"labels":[],"label_agreement":null},{"id":"W2032328503","doi":"10.1007/s10791-009-9108-x","title":"Document clustering of scientific texts using citation contexts","year":2009,"lang":"en","type":"article","venue":"Information Retrieval","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Cluster analysis; Information retrieval; Computer science; Document clustering; Citation; Vocabulary; Context (archaeology); Similarity (geometry); Representation (politics); Document retrieval; Natural language processing; Artificial intelligence; World Wide Web; Linguistics","score_opus":0.016025685924846902,"score_gpt":0.2950436954188471,"score_spread":0.27901800949400024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032328503","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43528467,0.02988485,0.488127,0.002138877,0.0019418302,0.0011894114,0.009118278,0.005384562,0.026930515],"genre_scores_gemma":[0.6016658,0.006460908,0.36630785,0.00015246746,0.0016504092,0.0005931752,0.01205742,0.0006688847,0.010443041],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978617,0.00055193325,0.00030844708,0.00042824374,0.00069155026,0.00015813031],"domain_scores_gemma":[0.99230534,0.0035796973,0.00060034555,0.0005478903,0.0026807154,0.0002859683],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0016699878,0.0008086043,0.0011066889,0.02648126,0.0024123208,0.0040888195,0.0010495521,0.0011271278,0.0031576708],"category_scores_gemma":[0.011847319,0.00038917956,0.0012473554,0.022339862,0.00064202555,0.0024870087,0.0012086096,0.0009756474,0.0020975948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008256128,0.00041090476,0.015022291,0.0016576733,0.00042566075,0.00037108985,0.0013423082,0.010132029,0.037552208,0.013687033,0.017721085,0.900852],"study_design_scores_gemma":[0.00044168904,0.0011516517,0.10011501,0.0012248359,0.003434505,0.0022770888,0.0038670625,0.50720596,0.103658564,0.12631258,0.14975165,0.00055942923],"about_ca_topic_score_codex":0.00395782,"about_ca_topic_score_gemma":0.008434736,"teacher_disagreement_score":0.9735187,"about_ca_system_score_codex":0.0010436877,"about_ca_system_score_gemma":0.002374801,"threshold_uncertainty_score":0.010563493},"labels":[],"label_agreement":null},{"id":"W2033153019","doi":"10.1207/s15327663jcp1502_5","title":"When Categorization Is Ambiguous: Factors That Facilitate the Use of a Multiple Category Inference Strategy","year":2005,"lang":"en","type":"article","venue":"Journal of Consumer Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Categorization; Cued speech; Ambiguity; Inference; Psychology; Cognitive psychology; Product category; Concept learning; Perception; Product (mathematics); Artificial intelligence; Computer science; Mathematics","score_opus":0.14715482868470045,"score_gpt":0.3529718958089943,"score_spread":0.20581706712429385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2033153019","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9798293,0.0002677387,0.008055402,0.00046688184,0.000025045245,0.00014716975,0.00003831033,0.000088035595,0.011082127],"genre_scores_gemma":[0.9918622,0.00013068004,0.0069866288,0.00014606606,0.000029634397,0.00004428036,0.00006864114,0.000059960832,0.00067194813],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9952154,0.0014716473,0.00051037816,0.000802897,0.0015775901,0.00042206247],"domain_scores_gemma":[0.92055416,0.0543511,0.012694224,0.0061405073,0.0042812265,0.0019787538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073771826,0.00045322545,0.0005878312,0.0010667101,0.0007849711,0.0043700165,0.0009964089,0.0016163359,0.0066807843],"category_scores_gemma":[0.10920677,0.000704266,0.0004819829,0.0006338354,0.0017454312,0.004967027,0.0020548697,0.0017785947,0.0007938524],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007005037,0.0020304269,0.49037546,0.0012789486,0.00053214794,0.0065526925,0.044287056,0.0045532463,0.23798758,0.024517905,0.0022362769,0.17864324],"study_design_scores_gemma":[0.0005274292,0.0017931157,0.85626805,0.00033470735,0.00054774963,0.0052893586,0.012951393,0.034064632,0.02790362,0.052200314,0.007710539,0.00040910335],"about_ca_topic_score_codex":0.002202364,"about_ca_topic_score_gemma":0.002375506,"teacher_disagreement_score":0.0073771826,"about_ca_system_score_codex":0.0006655748,"about_ca_system_score_gemma":0.0010697438,"threshold_uncertainty_score":0.039014697},"labels":[],"label_agreement":null},{"id":"W2033962915","doi":"10.1145/900051.900086","title":"User-controlled link adaptation","year":2003,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Adaptive hypermedia; Computer science; Adaptation (eye); Annotation; Hypermedia; Focus (optics); Task (project management); Human–computer interaction; Link (geometry); Adaptive system; User modeling; Architecture; Control (management); Information retrieval; World Wide Web; User interface; Artificial intelligence","score_opus":0.013951534834722477,"score_gpt":0.2638670876887624,"score_spread":0.2499155528540399,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2033962915","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047509395,0.00038870002,0.8981643,0.0002319839,0.00012430603,0.0004783436,0.00020054373,0.03608537,0.016817154],"genre_scores_gemma":[0.5053506,0.00058874453,0.45368505,0.0007552887,0.0002166133,0.0011835955,0.0008897274,0.0050606625,0.032269653],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99712974,0.0010009095,0.00017243564,0.0006577121,0.0009160753,0.00012306591],"domain_scores_gemma":[0.98731184,0.0060193497,0.00041829952,0.0045573637,0.0014028668,0.0002903102],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028313703,0.0010207575,0.0006410491,0.0011402251,0.0008222861,0.002844047,0.0028138864,0.001450436,0.008455289],"category_scores_gemma":[0.018469162,0.0005871942,0.0005723133,0.0010434804,0.0009481117,0.004839015,0.0032329562,0.0018762104,0.0031212263],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010637336,0.0007786829,0.0046356637,0.00062214804,0.00014458464,0.0008298076,0.0059949704,0.008047794,0.21045561,0.024622455,0.011456124,0.7313484],"study_design_scores_gemma":[0.00027391888,0.0008385391,0.010186314,0.00028356432,0.00038420258,0.0022724348,0.0014341582,0.27616322,0.2934691,0.05029327,0.36379963,0.0006017191],"about_ca_topic_score_codex":0.0008207537,"about_ca_topic_score_gemma":0.0010260468,"teacher_disagreement_score":0.008455289,"about_ca_system_score_codex":0.00032637076,"about_ca_system_score_gemma":0.00055099593,"threshold_uncertainty_score":0.028285742},"labels":[],"label_agreement":null},{"id":"W2035069493","doi":"10.1109/ichit.2006.165","title":"Keyword Extraction from Documents Using a Neural Network Model","year":2006,"lang":"en","type":"article","venue":"International Conference on Hybrid Information Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Artificial neural network; Backpropagation; Keyword extraction; Selection (genetic algorithm); Word (group theory); Artificial intelligence; tf–idf; Feature selection; Natural language processing; Feature (linguistics); Feature extraction; Information retrieval; Term (time); Mathematics","score_opus":0.02295526057919053,"score_gpt":0.30580536325677865,"score_spread":0.28285010267758814,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2035069493","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047931015,0.0013120198,0.9453168,0.00044641184,0.00013125084,0.00021635457,0.0003960742,0.0020478582,0.0022022375],"genre_scores_gemma":[0.45333624,0.00202038,0.53195727,0.00026389814,0.0002525121,0.00052642776,0.0012547348,0.00011219511,0.010276292],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995184,0.00009004165,0.00006833291,0.00011878009,0.00015854836,0.00004592952],"domain_scores_gemma":[0.9990345,0.0005604355,0.00007518344,0.000041834952,0.00027021996,0.000017900325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006860266,0.0007294933,0.00095416204,0.0014957372,0.0003195504,0.001224596,0.00089916575,0.0011148747,0.0013927614],"category_scores_gemma":[0.0033638314,0.00042304408,0.0007056083,0.0021357972,0.00023501455,0.0019152262,0.00041395432,0.0007761783,0.0011241067],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005341566,0.00020191348,0.002418849,0.00051207055,0.00016314069,0.00038964744,0.00018121481,0.3126332,0.02311178,0.004130106,0.0036658086,0.65205806],"study_design_scores_gemma":[0.000010835807,0.000025876367,0.00023649675,0.00001088544,0.00001611408,0.000048784506,0.000012084044,0.9952478,0.0023554396,0.0014012826,0.00062685786,0.000007579604],"about_ca_topic_score_codex":0.0061274753,"about_ca_topic_score_gemma":0.0055699092,"teacher_disagreement_score":0.0061274753,"about_ca_system_score_codex":0.00082029076,"about_ca_system_score_gemma":0.00085466396,"threshold_uncertainty_score":0.012183607},"labels":[],"label_agreement":null},{"id":"W2037337707","doi":"10.1145/1458082.1458140","title":"Relating dependent indexes using dempster-shafer theory","year":2008,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Search engine indexing; Dempster–Shafer theory; Computer science; Term (time); Information retrieval; Artificial intelligence; Index (typography); Data mining; World Wide Web","score_opus":0.036428326711995765,"score_gpt":0.2891301163310142,"score_spread":0.25270178961901846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2037337707","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017412439,0.0038896613,0.94364506,0.0007787826,0.00034493764,0.00033440904,0.0005843,0.00043048384,0.03257993],"genre_scores_gemma":[0.5248579,0.012098774,0.42263693,0.00076624786,0.0013150572,0.00090950937,0.0028859514,0.0007296048,0.033800166],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9921355,0.0020585242,0.000637879,0.0010967535,0.0037358815,0.00033550579],"domain_scores_gemma":[0.9779704,0.012339227,0.0013845841,0.0044451603,0.0035304266,0.00033029137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065224376,0.001563802,0.0023920643,0.012695759,0.0026500758,0.0073233778,0.0041088895,0.0038725478,0.010940888],"category_scores_gemma":[0.056183644,0.0010382438,0.0019543013,0.01654731,0.0028644444,0.018652735,0.0052091256,0.00405346,0.004791971],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007180848,0.00006431127,0.00085004506,0.00024403652,0.00010462071,0.00022437725,0.0003682493,0.027732164,0.00083819946,0.90282,0.0041571474,0.0625249],"study_design_scores_gemma":[0.000016629881,0.000027035758,0.00040960766,0.000049405968,0.000052476695,0.00018990012,0.000079827696,0.1128735,0.0011327007,0.8759737,0.009133605,0.00006156594],"about_ca_topic_score_codex":0.0022644515,"about_ca_topic_score_gemma":0.0012926802,"teacher_disagreement_score":0.012695759,"about_ca_system_score_codex":0.003935607,"about_ca_system_score_gemma":0.0018409113,"threshold_uncertainty_score":0.036600888},"labels":[],"label_agreement":null},{"id":"W2037847309","doi":"10.1109/hicss.2014.228","title":"Introduction to Text Analytics Minitrack","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Analytics; Data science","score_opus":0.009641329039161796,"score_gpt":0.2665987821706924,"score_spread":0.25695745313153057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2037847309","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033243408,0.03686497,0.58431154,0.029918171,0.04503751,0.002827648,0.06404933,0.06967258,0.16399388],"genre_scores_gemma":[0.012328551,0.034113474,0.41774788,0.013817848,0.035663404,0.002491752,0.120239325,0.021886928,0.34171084],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9962484,0.00053738616,0.0003908272,0.0007073495,0.0018655239,0.00025051675],"domain_scores_gemma":[0.9777909,0.0073782606,0.0009789332,0.00225266,0.009854972,0.0017443434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003908742,0.0017810405,0.0014764736,0.010258445,0.0022683784,0.010410162,0.0029030032,0.0022434539,0.18268302],"category_scores_gemma":[0.021466874,0.0011954858,0.0017235407,0.009391974,0.0010217977,0.014142364,0.0060864063,0.004833133,0.20301957],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000045788933,0.000066126864,0.0003339064,0.0009350448,0.00002672037,0.00008068869,0.00015241145,0.00020354877,0.0036651136,0.006812019,0.67249376,0.31518483],"study_design_scores_gemma":[0.000008813001,0.000027390033,0.0005060645,0.00030443823,0.000013842945,0.00023324843,0.0001010925,0.000734854,0.0019500488,0.007303679,0.988785,0.000031489588],"about_ca_topic_score_codex":0.002017602,"about_ca_topic_score_gemma":0.0029904381,"teacher_disagreement_score":0.18268302,"about_ca_system_score_codex":0.0015565057,"about_ca_system_score_gemma":0.0032902465,"threshold_uncertainty_score":0.6111356},"labels":[],"label_agreement":null},{"id":"W2039979754","doi":"10.1109/isspa.2012.6310552","title":"Automatic Document Topic Identification using Wikipedia Hierarchical Ontology","year":2012,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Ontology; Information retrieval; Identification (biology); Document clustering; Cluster analysis; Construct (python library); Hierarchical clustering; Data mining; Artificial intelligence","score_opus":0.02265378680570878,"score_gpt":0.3276773295832177,"score_spread":0.3050235427775089,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2039979754","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1675846,0.001452167,0.80934465,0.00044846392,0.00014533232,0.00066975004,0.004728595,0.0070249434,0.00860152],"genre_scores_gemma":[0.38071784,0.0010050873,0.60277075,0.00007278266,0.00007142871,0.00037666518,0.011460164,0.0003175292,0.0032077865],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99912816,0.0001785933,0.00008162215,0.00024016981,0.00027503484,0.00009647405],"domain_scores_gemma":[0.9985238,0.0004957066,0.00021553325,0.00015943767,0.0005203333,0.00008522957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00094353204,0.00054783165,0.0004775235,0.0068765613,0.0009907448,0.0016111074,0.00065658306,0.0006446639,0.00092289614],"category_scores_gemma":[0.0034320415,0.00024860693,0.0008291526,0.00403108,0.00030034647,0.002627577,0.0011735952,0.00067063747,0.0007611985],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005147996,0.00039445484,0.020643378,0.00085554074,0.00023813582,0.00064939074,0.002412258,0.013070612,0.073247194,0.016912416,0.025351608,0.8457102],"study_design_scores_gemma":[0.00011202786,0.00015494456,0.032195333,0.00028688458,0.0003771449,0.0011332327,0.0033015867,0.76607955,0.079575986,0.030572085,0.08601855,0.00019266288],"about_ca_topic_score_codex":0.014828493,"about_ca_topic_score_gemma":0.016419142,"teacher_disagreement_score":0.014828493,"about_ca_system_score_codex":0.000862593,"about_ca_system_score_gemma":0.0021375085,"threshold_uncertainty_score":0.029484391},"labels":[],"label_agreement":null},{"id":"W2040027527","doi":"10.3758/s13423-015-0808-5","title":"A rational model of function learning","year":2015,"lang":"en","type":"review","venue":"Psychonomic Bulletin & Review","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":112,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Air Force Office of Scientific Research; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Similarity (geometry); Associative learning; Associative property; Psychology; Artificial intelligence; Function (biology); Equivalence (formal languages); Probabilistic logic; Regression; Machine learning; Rational function; Regression analysis; Computer science; Cognitive psychology; Mathematics","score_opus":0.08088215470050462,"score_gpt":0.36584288331871384,"score_spread":0.2849607286182092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2040027527","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019858941,0.0921886,0.73557115,0.02568218,0.00081449293,0.00013139621,0.00046199936,0.0008022142,0.12448905],"genre_scores_gemma":[0.7279528,0.1007966,0.10906463,0.00452362,0.0014256231,0.0004168244,0.00062710274,0.00017048996,0.055022415],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99938834,0.00021640376,0.00003192403,0.00012058049,0.00017450014,0.00006824236],"domain_scores_gemma":[0.9990375,0.0006032981,0.00006905994,0.00010032057,0.00015921815,0.00003060807],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013561544,0.00072878523,0.0007792951,0.00097017526,0.0003627745,0.0018073139,0.0013167885,0.0013316714,0.006045928],"category_scores_gemma":[0.0034239062,0.00021022759,0.00069335033,0.00087550184,0.002994275,0.003156775,0.00074703526,0.0014910828,0.0021588686],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030753974,0.00002288892,0.00029204614,0.0002912593,0.000030220086,0.00004734898,0.000056648943,0.006718756,0.0002737608,0.8678882,0.005304125,0.119043946],"study_design_scores_gemma":[0.000017033619,0.000025537443,0.00024374912,0.00011496964,0.000019672192,0.000074059775,0.000026597978,0.016145097,0.0002737497,0.95913917,0.023908816,0.000011493329],"about_ca_topic_score_codex":0.002893114,"about_ca_topic_score_gemma":0.0012287755,"teacher_disagreement_score":0.006045928,"about_ca_system_score_codex":0.0016835252,"about_ca_system_score_gemma":0.0019107169,"threshold_uncertainty_score":0.020225704},"labels":[],"label_agreement":null},{"id":"W2052799227","doi":"10.1109/3pgcic.2013.68","title":"Principal Component Analysis in Business Intelligence Applications","year":2013,"lang":"en","type":"article","venue":"2013 Eighth International Conference on P2P, Parallel, Grid, Cloud and Internet Computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Competitor analysis; Business intelligence; Context (archaeology); Identification (biology); Task (project management); Principal component analysis; Component (thermodynamics); Domain (mathematical analysis); Data science; Principal (computer security); Big data; Market intelligence; Business information; Chart; Competitive intelligence; Information retrieval; Artificial intelligence; Data mining; Knowledge management; Engineering; Marketing; Business; Computer security","score_opus":0.029996336928425234,"score_gpt":0.29628025998283475,"score_spread":0.2662839230544095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052799227","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005278931,0.008506055,0.97238696,0.0017488004,0.00027156394,0.00027245394,0.000419314,0.0020382062,0.009077855],"genre_scores_gemma":[0.12972385,0.012621959,0.84808755,0.00048610405,0.00066087075,0.00077997445,0.0011975168,0.0005203969,0.005921717],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9959282,0.0018530143,0.0003474899,0.0006127154,0.0011316107,0.00012694467],"domain_scores_gemma":[0.9941614,0.003603389,0.0004500404,0.00053971534,0.0011253646,0.00012008021],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036516732,0.0016059636,0.001174283,0.0037708587,0.0008943698,0.0035541332,0.001000207,0.0016868422,0.00495576],"category_scores_gemma":[0.015943322,0.0007709122,0.0011881449,0.008625197,0.0012095834,0.0023507355,0.001628162,0.0025216676,0.00399412],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015365539,0.00023139949,0.0052351756,0.0013486047,0.0003549571,0.0004967678,0.0008568668,0.08259351,0.0050257565,0.11670108,0.029519044,0.7574831],"study_design_scores_gemma":[0.000048252827,0.000118078875,0.0074374145,0.0005132998,0.0001337602,0.0004279926,0.00054024963,0.59156764,0.0043599764,0.29992986,0.09476289,0.0001606712],"about_ca_topic_score_codex":0.00532094,"about_ca_topic_score_gemma":0.0030446632,"teacher_disagreement_score":0.00532094,"about_ca_system_score_codex":0.00097529276,"about_ca_system_score_gemma":0.0018373942,"threshold_uncertainty_score":0.019312143},"labels":[],"label_agreement":null},{"id":"W2053624351","doi":"10.1111/j.0037-976x.2003.00263.x","title":"III. Study 2: Rule Complexity and Stimulus Characteristics in Executive Function","year":2003,"lang":"en","type":"article","venue":"Monographs of the Society for Research in Child Development","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; University of Toronto","funders":"","keywords":"Psychology; Executive functions; Cognitive psychology; Stimulus (psychology); Cognition; Developmental psychology; Neuroscience","score_opus":0.08148596802906449,"score_gpt":0.36731655988695855,"score_spread":0.28583059185789406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053624351","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9936103,0.00030958492,0.0005432587,0.00028975564,0.00008950157,0.00032292685,0.00060155144,0.000025843796,0.004207294],"genre_scores_gemma":[0.9840661,0.00020168992,0.0014277333,0.00087055005,0.00013889352,0.0009684862,0.0015003025,0.00008705355,0.01073915],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99833554,0.0004307225,0.00018044584,0.0005732247,0.0002647684,0.0002152632],"domain_scores_gemma":[0.9697506,0.02302344,0.002645021,0.0026626561,0.0007218225,0.0011965373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005821828,0.0018372926,0.0022868803,0.0008540786,0.0014323109,0.0050385343,0.00244186,0.0040327176,0.024208555],"category_scores_gemma":[0.025087519,0.0015639502,0.0018618903,0.0005496539,0.0014745108,0.007097481,0.002442763,0.0053875386,0.0042835907],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.121103406,0.041209433,0.56667536,0.003207068,0.0043190257,0.0031658495,0.034620333,0.005548053,0.1372382,0.012571086,0.009419106,0.060922958],"study_design_scores_gemma":[0.01153711,0.026204742,0.898491,0.00040170984,0.0017160793,0.0025210443,0.0075949146,0.009293866,0.013989617,0.01897183,0.008965572,0.00031234932],"about_ca_topic_score_codex":0.0022092925,"about_ca_topic_score_gemma":0.0023226142,"teacher_disagreement_score":0.024208555,"about_ca_system_score_codex":0.0006871148,"about_ca_system_score_gemma":0.0010945576,"threshold_uncertainty_score":0.080985665},"labels":[],"label_agreement":null},{"id":"W2056425533","doi":"10.1167/8.6.628","title":"Visual spread reading: Noisy letters in their natural context","year":2010,"lang":"en","type":"article","venue":"Journal of Vision","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Legibility; Noise (video); Computer science; Context (archaeology); Artificial intelligence; Reading (process); Natural language processing; Natural (archaeology); Stimulus (psychology); Speech recognition; Pattern recognition (psychology); Psychology; Cognitive psychology; Computer vision; Communication; Image (mathematics); Linguistics; Geography; Art; Visual arts","score_opus":0.007233417522279048,"score_gpt":0.30445572434014584,"score_spread":0.29722230681786677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056425533","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99164563,0.0001369643,0.006598958,0.000014275602,0.0000069443536,0.000033692184,0.0000643322,0.00007589539,0.0014232883],"genre_scores_gemma":[0.9969548,0.000024571153,0.0024842815,0.000020414756,0.000007883871,0.000027126396,0.00006567108,0.000026522841,0.00038873282],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9986533,0.00039281123,0.00011087614,0.00040396053,0.00039808534,0.000041007042],"domain_scores_gemma":[0.9886001,0.0069019645,0.002309809,0.0011239882,0.0006515321,0.00041258597],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012588486,0.00046138876,0.00032767674,0.0009541392,0.00028293114,0.0010862104,0.0003545129,0.0005493192,0.0024191306],"category_scores_gemma":[0.015156264,0.0002667552,0.00015083064,0.0004529592,0.0008691338,0.0014177245,0.001114624,0.000388399,0.00034804488],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004508784,0.0005121249,0.111918114,0.0008292073,0.00016827452,0.0011934779,0.0066809156,0.003727638,0.8126648,0.0016458845,0.0005364078,0.055614453],"study_design_scores_gemma":[0.00008246288,0.0021272337,0.8246888,0.000100704245,0.0001310017,0.0023628785,0.0014076094,0.013153563,0.15005194,0.0040067392,0.0017841011,0.00010297239],"about_ca_topic_score_codex":0.0005530727,"about_ca_topic_score_gemma":0.0006923342,"teacher_disagreement_score":0.0024191306,"about_ca_system_score_codex":0.00029463784,"about_ca_system_score_gemma":0.00011410301,"threshold_uncertainty_score":0.008092821},"labels":[],"label_agreement":null},{"id":"W2057505677","doi":"10.1111/j.1467-8640.2007.00293.x","title":"KEYWORD EXTRACTION STRATEGY FOR ITEM BANKS TEXT CATEGORIZATION","year":2007,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Categorization; Computer science; Selection (genetic algorithm); Keyword extraction; Natural language processing; Phrase; Sentence; Artificial intelligence; Text categorization; Feature selection; Information retrieval","score_opus":0.04857456591788745,"score_gpt":0.37604187604377143,"score_spread":0.327467310125884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2057505677","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028146092,0.00063242146,0.9650881,0.00022884634,0.00009358824,0.0007972686,0.0009756659,0.0028843698,0.0011535421],"genre_scores_gemma":[0.15709251,0.00033176132,0.8361358,0.00020640905,0.000110476685,0.0012688038,0.0027596557,0.00019002566,0.0019046229],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997898,0.0005709974,0.0003870305,0.00044564888,0.00056656415,0.0001318175],"domain_scores_gemma":[0.9973259,0.00091957767,0.00019197867,0.0002405558,0.0012307309,0.00009118573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016465919,0.0014441381,0.0017516796,0.0049436893,0.0006792031,0.0011466541,0.0013045309,0.0010281429,0.0033725486],"category_scores_gemma":[0.0048685144,0.00046065505,0.0011255377,0.004691612,0.00044951765,0.0023939824,0.0008494055,0.00077434105,0.0032637739],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042625834,0.0002469761,0.0029040033,0.000748839,0.00019192301,0.00043398084,0.000601656,0.0028805165,0.08757214,0.003694803,0.0067995396,0.8934994],"study_design_scores_gemma":[0.000566257,0.002014619,0.023180556,0.00025905066,0.0010419507,0.004101768,0.002232263,0.61095023,0.24058695,0.04175271,0.07284662,0.00046709902],"about_ca_topic_score_codex":0.0013121639,"about_ca_topic_score_gemma":0.0014213767,"teacher_disagreement_score":0.0049436893,"about_ca_system_score_codex":0.00051211094,"about_ca_system_score_gemma":0.001059204,"threshold_uncertainty_score":0.011282325},"labels":[],"label_agreement":null},{"id":"W2061449990","doi":"10.1002/asi.1101.abs","title":"User preferences in the classification of electronic bookmarks: Implications for a shared system","year":2001,"lang":"en","type":"article","venue":"Journal of the American Society for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Categorization; Computer science; Context (archaeology); World Wide Web; The Internet; Documentation; Information retrieval; Artificial intelligence","score_opus":0.018552468412853438,"score_gpt":0.3082505032618173,"score_spread":0.28969803484896384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061449990","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99405825,0.000027870552,0.002940112,0.00033856087,0.00000394577,0.00004270545,0.000013276372,0.000019961028,0.0025554113],"genre_scores_gemma":[0.997242,0.000018348863,0.0023086758,0.00004891042,0.000004471723,0.000025376135,0.000017234464,0.0000076181063,0.00032734772],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.9874817,0.008221757,0.0006833698,0.0009421365,0.0020013538,0.0006696801],"domain_scores_gemma":[0.9487753,0.036637086,0.0039920686,0.0035869663,0.0044506188,0.0025579957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012772173,0.00021902399,0.00033100208,0.0011043851,0.0018744031,0.004696914,0.0006276512,0.00096963695,0.0026198195],"category_scores_gemma":[0.053662688,0.00028139295,0.0004220846,0.000961336,0.0017747198,0.005395834,0.0018721732,0.0007728132,0.00044465752],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012232574,0.0010805853,0.6328707,0.00018471065,0.000082825434,0.0007115511,0.1875381,0.0010346276,0.013784688,0.006174813,0.0009696545,0.15434457],"study_design_scores_gemma":[0.00010818886,0.0019815455,0.6028893,0.00016777693,0.00009594376,0.0015572022,0.34802577,0.018485146,0.006351231,0.011544653,0.008553455,0.00023985472],"about_ca_topic_score_codex":0.0029029287,"about_ca_topic_score_gemma":0.0034851544,"teacher_disagreement_score":0.012772173,"about_ca_system_score_codex":0.0007572006,"about_ca_system_score_gemma":0.00077294285,"threshold_uncertainty_score":0.06754649},"labels":[],"label_agreement":null},{"id":"W2061552568","doi":"10.1037/0278-7393.32.6.1431","title":"Relation availability was not confounded with familiarity or plausibility in Gagné and Shoben (1997): Comment on Wisniewski and Murphy (2005).","year":2006,"lang":"en","type":"letter","venue":"Journal of Experimental Psychology Learning Memory and Cognition","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"National Institute of Mental Health","keywords":"Relation (database); Phrase; Psychology; Artificial intelligence; Computer science; Data mining","score_opus":0.03262812095320073,"score_gpt":0.32581819497411074,"score_spread":0.29319007402091,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061552568","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015031125,0.0046284893,0.0058558313,0.925224,0.02751398,0.000102581456,0.000952234,0.00054376916,0.020147933],"genre_scores_gemma":[0.23852113,0.0032947683,0.009410393,0.6714684,0.034307186,0.00035377784,0.00034399424,0.00033578326,0.041964628],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99585634,0.00061365706,0.000486678,0.0009015221,0.0018610596,0.0002806126],"domain_scores_gemma":[0.96123165,0.022568982,0.0027105256,0.0017566467,0.01097887,0.00075330806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00621471,0.0008000039,0.000875879,0.0007909562,0.0029077192,0.0024461043,0.004149344,0.013544509,0.008949578],"category_scores_gemma":[0.04094341,0.0005427853,0.0008290688,0.00056782877,0.0029116177,0.0041248617,0.0017589001,0.009578103,0.005872984],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010736369,0.00030424603,0.011043515,0.0005573234,0.00006958173,0.0031066362,0.0052004266,0.00030674678,0.005974442,0.04745157,0.88283116,0.042080652],"study_design_scores_gemma":[0.0003741973,0.00047446913,0.09321195,0.0006461148,0.00022040511,0.005717108,0.004326449,0.0039566914,0.016676541,0.094160035,0.7797734,0.00046262975],"about_ca_topic_score_codex":0.017658817,"about_ca_topic_score_gemma":0.019914009,"teacher_disagreement_score":0.017658817,"about_ca_system_score_codex":0.0027853823,"about_ca_system_score_gemma":0.0015989847,"threshold_uncertainty_score":0.035112083},"labels":[],"label_agreement":null},{"id":"W2066433987","doi":"10.1371/journal.pone.0071914","title":"Connected Text Reading and Differences in Text Reading Fluency in Adult Readers","year":2013,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"FP7 People: Marie-Curie Actions; National Science Foundation","keywords":"Reading (process); Fluency; Computer science; Cognitive psychology; Cognition; Reading comprehension; Psychology; Linguistics; Artificial intelligence; Mathematics education","score_opus":0.031040022811816538,"score_gpt":0.23764317546717234,"score_spread":0.2066031526553558,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2066433987","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99935466,0.000043296466,0.000098361794,0.000007668811,0.0000017867704,0.000003972813,0.000031021213,0.000007038015,0.00045224163],"genre_scores_gemma":[0.9994081,0.000020490284,0.000102251215,0.0000069175253,0.0000043363225,0.000006070737,0.00005374184,0.000004354212,0.00039371705],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99959964,0.000073499476,0.00006945159,0.00012832467,0.00008331632,0.00004581469],"domain_scores_gemma":[0.991397,0.004785871,0.0022200772,0.000408766,0.00058347476,0.00060471933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007329502,0.00024111115,0.00026426278,0.0013845459,0.00017244508,0.0010469027,0.00019471098,0.0006539963,0.0040987143],"category_scores_gemma":[0.009643594,0.00014646145,0.00017014232,0.00038743383,0.00053975894,0.0011248324,0.00051482336,0.00033366377,0.0007546548],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022724655,0.0006284341,0.924098,0.0001084989,0.00013406991,0.00079161243,0.0139504345,0.00022494984,0.026544465,0.00022942136,0.00023538688,0.030782131],"study_design_scores_gemma":[0.000009419763,0.0004964896,0.9963929,0.0000055994115,0.000014699035,0.00039700288,0.0010650856,0.0002452469,0.0011396484,0.00015121636,0.000074448224,0.0000081149],"about_ca_topic_score_codex":0.0006621299,"about_ca_topic_score_gemma":0.0008014808,"teacher_disagreement_score":0.0040987143,"about_ca_system_score_codex":0.00012830844,"about_ca_system_score_gemma":0.00006012833,"threshold_uncertainty_score":0.013711572},"labels":[],"label_agreement":null},{"id":"W206788155","doi":"10.4018/978-1-60566-904-5.ch015","title":"Using Graphics to Improve Understanding of Conceptual Models","year":2010,"lang":"en","type":"book-chapter","venue":"Advances in database research (ADR) book series/Advances in database research series","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Graphics; Cognitive load; Computer science; Comprehension; Cognition; Domain (mathematical analysis); Human–computer interaction; Computer graphics; Multimedia; Artificial intelligence; Computer graphics (images); Programming language; Psychology","score_opus":0.20224303746265215,"score_gpt":0.4459812781824064,"score_spread":0.24373824071975425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W206788155","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22263326,0.008497502,0.6109631,0.0035238822,0.00045148958,0.00038671575,0.00075596577,0.012686135,0.1401019],"genre_scores_gemma":[0.35106018,0.008061513,0.6085919,0.00060561736,0.00015763,0.00024451164,0.0014231743,0.0009476055,0.028907903],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996314,0.0001564585,0.000017950248,0.000069899404,0.000104518454,0.000019677336],"domain_scores_gemma":[0.99562883,0.0035003223,0.00020555416,0.0003854721,0.00021192527,0.00006781314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007737963,0.0010454208,0.0002976033,0.00110835,0.00013685849,0.0021565957,0.0009790667,0.00068163243,0.016780455],"category_scores_gemma":[0.007450213,0.00023256506,0.00048081047,0.0011076228,0.00048614395,0.0038681796,0.000984701,0.0011161813,0.002862428],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015064335,0.00026910868,0.0018449213,0.00095170125,0.000035777382,0.00017428533,0.0030723417,0.0081616435,0.028310705,0.022950869,0.01873655,0.91534144],"study_design_scores_gemma":[0.00031898625,0.0019159612,0.02383504,0.0013775234,0.00034702063,0.0025597252,0.003798313,0.10884729,0.07565531,0.13955759,0.64152044,0.00026681193],"about_ca_topic_score_codex":0.00044436744,"about_ca_topic_score_gemma":0.0005572857,"teacher_disagreement_score":0.016780455,"about_ca_system_score_codex":0.00041821264,"about_ca_system_score_gemma":0.0003205729,"threshold_uncertainty_score":0.05613625},"labels":[],"label_agreement":null},{"id":"W2071049913","doi":"10.3390/informatics1010032","title":"Using Collaborative Tagging for Text Classification: From Text Classification to Opinion Mining","year":2013,"lang":"en","type":"article","venue":"Informatics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Polytechnique Montréal","funders":"Génome Québec; Genome Canada","keywords":"Ranking (information retrieval); Computer science; Task (project management); Context (archaeology); Information retrieval; Document classification; Natural language processing; Artificial intelligence; World Wide Web; Engineering; Geography","score_opus":0.07513631016913433,"score_gpt":0.34807593113271956,"score_spread":0.27293962096358526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2071049913","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02970009,0.0012694129,0.9547403,0.0012152162,0.00046975765,0.0007134723,0.0011117249,0.004134496,0.0066455444],"genre_scores_gemma":[0.2364402,0.000949974,0.75347817,0.0004994478,0.0009447619,0.0007312197,0.0030209739,0.00029994705,0.0036353092],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99379915,0.0026210335,0.0005609248,0.001215105,0.0014594845,0.00034435638],"domain_scores_gemma":[0.98388493,0.009603442,0.0015055583,0.0015487162,0.002976781,0.00048065354],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062494655,0.0018831796,0.0018036489,0.007568046,0.0013643074,0.0045530675,0.002479907,0.0022713481,0.0024648705],"category_scores_gemma":[0.019560594,0.00050571776,0.0016387329,0.005832557,0.00093346316,0.0057226475,0.0021496436,0.0023706283,0.0044026994],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022333156,0.0006022223,0.007998455,0.0005924674,0.00022386294,0.00027690057,0.0014701571,0.0048284815,0.018083243,0.0038468023,0.014358052,0.94749606],"study_design_scores_gemma":[0.0001447746,0.0005865489,0.015677555,0.0005214504,0.00050790026,0.00091843033,0.003267175,0.79065156,0.0545004,0.08509315,0.047798198,0.00033282145],"about_ca_topic_score_codex":0.0023208207,"about_ca_topic_score_gemma":0.003110581,"teacher_disagreement_score":0.007568046,"about_ca_system_score_codex":0.001023664,"about_ca_system_score_gemma":0.0010555263,"threshold_uncertainty_score":0.033050776},"labels":[],"label_agreement":null},{"id":"W2072435181","doi":"10.1207/s1532690xci2402_3","title":"Helping Students Understand Challenging Topics in Science Through Ontology Training","year":2006,"lang":"en","type":"article","venue":"Cognition and Instruction","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":285,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Spencer Foundation; Andrew W. Mellon Foundation","keywords":"Ontology; Conceptual change; Science education; Task (project management); Mathematics education; Computer science; Concept learning; Psychology; Epistemology; Engineering","score_opus":0.04250187749906544,"score_gpt":0.3179108858070179,"score_spread":0.27540900830795245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2072435181","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9336768,0.00020467957,0.051261306,0.002379022,0.00007126378,0.00045175647,0.000074875345,0.0007500224,0.011130246],"genre_scores_gemma":[0.8601751,0.0007236146,0.12945162,0.000835301,0.000044299595,0.00052988966,0.00032655173,0.00008277627,0.007830862],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9992399,0.00023774903,0.000051204657,0.00016622411,0.00018560997,0.000119256896],"domain_scores_gemma":[0.99455845,0.0032982551,0.00064296386,0.00051411113,0.00043095023,0.0005552832],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019198084,0.00073743175,0.00042846895,0.0010300237,0.00069514563,0.0018849963,0.0010286861,0.0010015187,0.0051772906],"category_scores_gemma":[0.012199418,0.00026286187,0.000455368,0.00050522346,0.00090522476,0.0035665517,0.0024688141,0.0015471035,0.0013882688],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004180742,0.013631689,0.046858,0.0010482885,0.000049747563,0.0010580041,0.06783462,0.0035147795,0.091381125,0.017000262,0.01744765,0.73975796],"study_design_scores_gemma":[0.0010124384,0.0071680546,0.12922157,0.0010805317,0.00045025357,0.004632069,0.07682286,0.06989305,0.16591468,0.22413993,0.3192632,0.00040133574],"about_ca_topic_score_codex":0.0005446271,"about_ca_topic_score_gemma":0.0011297136,"teacher_disagreement_score":0.0051772906,"about_ca_system_score_codex":0.0005128833,"about_ca_system_score_gemma":0.0012791485,"threshold_uncertainty_score":0.017319739},"labels":[],"label_agreement":null},{"id":"W2073654225","doi":"10.5430/air.v4n1p1","title":"Using a predefined passphrase to evaluate a speaker verification system","year":2014,"lang":"en","type":"article","venue":"Artificial Intelligence Research","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Speaker verification; Computer science; Biometrics; Variety (cybernetics); Process (computing); Feature (linguistics); Speaker recognition; Focus (optics); Speech recognition; Artificial intelligence; Natural language processing; Programming language; Linguistics","score_opus":0.32271136736696715,"score_gpt":0.4872202259271605,"score_spread":0.16450885856019337,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2073654225","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9211433,0.00051221857,0.06978961,0.000077117795,0.00019346722,0.0010416653,0.0014097844,0.0017904631,0.0040424084],"genre_scores_gemma":[0.9214254,0.00020410647,0.070850745,0.00008214763,0.00006347659,0.0009551624,0.0025590886,0.00019709648,0.0036627653],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9901322,0.0030598973,0.0012202346,0.0012893574,0.0038951728,0.00040307487],"domain_scores_gemma":[0.9786739,0.010847652,0.0020001563,0.0023452681,0.0056316233,0.00050146214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074616848,0.0015315083,0.0009936198,0.0017743508,0.0009133492,0.0015144278,0.0006681233,0.0013439283,0.0031135932],"category_scores_gemma":[0.024000138,0.00024225708,0.0007410633,0.00085539644,0.00071742653,0.001431186,0.0013152343,0.0006212483,0.0017881553],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008293963,0.0017853412,0.05400603,0.0019197266,0.0010761017,0.0007368829,0.002558165,0.012647905,0.4154055,0.0019034563,0.006078314,0.49358863],"study_design_scores_gemma":[0.000503068,0.024337059,0.26381356,0.00016296333,0.00065907027,0.002999665,0.0021445195,0.15776771,0.53315765,0.0017488342,0.012132542,0.0005734258],"about_ca_topic_score_codex":0.0013185178,"about_ca_topic_score_gemma":0.0016734065,"teacher_disagreement_score":0.0074616848,"about_ca_system_score_codex":0.00049137045,"about_ca_system_score_gemma":0.0005218969,"threshold_uncertainty_score":0.039461613},"labels":[],"label_agreement":null},{"id":"W2077409459","doi":"10.1504/ijamc.2008.018504","title":"Keyword extraction rules based on a part-of-speech hierarchy","year":2008,"lang":"en","type":"article","venue":"International Journal of Advanced Media and Communication","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Hierarchy; Natural language processing; Artificial intelligence; Sentence; Domain (mathematical analysis); Set (abstract data type); Field (mathematics); Context (archaeology); Natural language; Keyword extraction; Natural language understanding","score_opus":0.021887724369164682,"score_gpt":0.3129724551895625,"score_spread":0.2910847308203978,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077409459","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0171827,0.00039732043,0.97706443,0.00018374424,0.000058335383,0.00029651436,0.0005866194,0.0020803392,0.002150092],"genre_scores_gemma":[0.1671251,0.00045045262,0.82808954,0.00017730432,0.00006631409,0.00034850917,0.001484473,0.00020475834,0.0020535556],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980667,0.00033039085,0.00023763551,0.0005703054,0.0007012802,0.00009372838],"domain_scores_gemma":[0.9953511,0.00298692,0.00023980487,0.00031835434,0.0010072081,0.000096614494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015787024,0.00073482457,0.0009860562,0.0028644172,0.0007003997,0.0013822464,0.0015868827,0.00086662284,0.0026014922],"category_scores_gemma":[0.008465464,0.00047502946,0.0010441034,0.0014456476,0.0011237218,0.0025877934,0.00069628307,0.0011401523,0.0024858154],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000435098,0.00023160057,0.0064383936,0.0013723385,0.00020632148,0.0010767905,0.0015749678,0.049135502,0.096521944,0.02998577,0.00685038,0.806171],"study_design_scores_gemma":[0.00009433816,0.00030846673,0.0045476,0.00032081167,0.0003577867,0.0011924875,0.00046983708,0.7796551,0.10142568,0.08778316,0.023717383,0.00012738306],"about_ca_topic_score_codex":0.0024304178,"about_ca_topic_score_gemma":0.003037305,"teacher_disagreement_score":0.0028644172,"about_ca_system_score_codex":0.00059551856,"about_ca_system_score_gemma":0.0014904275,"threshold_uncertainty_score":0.008702874},"labels":[],"label_agreement":null},{"id":"W2081510074","doi":"10.4028/www.scientific.net/amr.760-762.852","title":"Semantic Similarity Measure Based on Concreteness Degree of a Concept","year":2013,"lang":"en","type":"article","venue":"Advanced materials research","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ministry of Transportation of Ontario","funders":"Division of Materials Research; Natural Science Foundation of Hebei Province","keywords":"Concreteness; Semantic similarity; Similarity (geometry); Process (computing); Measure (data warehouse); Task (project management); Degree (music); Computer science; Ontology; Semantics (computer science); Natural language processing; Artificial intelligence; Mathematics; Data mining; Image (mathematics); Cognitive psychology; Psychology; Engineering","score_opus":0.08846113571753662,"score_gpt":0.3716765264366494,"score_spread":0.2832153907191128,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2081510074","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2941103,0.0022796066,0.68323183,0.00037551828,0.0002453116,0.00050586136,0.0011026812,0.0008444775,0.017304389],"genre_scores_gemma":[0.82816875,0.00051981566,0.1689389,0.000047263646,0.00009118607,0.00028702457,0.00082482275,0.00005560856,0.0010666543],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9948744,0.00083786197,0.0007385705,0.0008073789,0.0026011472,0.00014067488],"domain_scores_gemma":[0.9934428,0.0027451706,0.0008321694,0.0006223223,0.002082064,0.00027546354],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021008733,0.0005306845,0.00065754243,0.008662741,0.0007057929,0.0016830862,0.0007043939,0.00067583995,0.0017945401],"category_scores_gemma":[0.013940509,0.00015995966,0.0008386144,0.0055661784,0.001162694,0.004556598,0.0012890465,0.000600258,0.0003390882],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008664565,0.00046118774,0.058767498,0.0021101963,0.0008698796,0.0004975779,0.0041581225,0.015870856,0.08040291,0.09830362,0.005783495,0.7319082],"study_design_scores_gemma":[0.00021256441,0.002091463,0.18804912,0.0005779069,0.0012204073,0.003895636,0.0060670925,0.36821324,0.1003702,0.2771062,0.051519193,0.0006770123],"about_ca_topic_score_codex":0.0009657568,"about_ca_topic_score_gemma":0.0009921107,"teacher_disagreement_score":0.008662741,"about_ca_system_score_codex":0.0010002671,"about_ca_system_score_gemma":0.0009021324,"threshold_uncertainty_score":0.011110604},"labels":[],"label_agreement":null},{"id":"W2084026949","doi":"10.1109/cec.2013.6557603","title":"An evolutionary algorithm for Feature Selective Double Clustering of text documents","year":2013,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Cluster analysis; Document clustering; Feature (linguistics); Artificial intelligence; Cluster (spacecraft); Data mining; Information retrieval; Algorithm; Pattern recognition (psychology)","score_opus":0.00884334672967403,"score_gpt":0.2884877279305971,"score_spread":0.2796443812009231,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084026949","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01826127,0.00020355596,0.9800724,0.00008565005,0.00003828462,0.000081047845,0.000043730746,0.00035575216,0.00085833087],"genre_scores_gemma":[0.14526892,0.00014323315,0.8510535,0.000102664846,0.000031703185,0.00029298195,0.00029426583,0.00010849403,0.0027041505],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99932027,0.0001243516,0.0000529098,0.00019621589,0.00024555493,0.00006072824],"domain_scores_gemma":[0.9991117,0.0004016444,0.000072417264,0.00008125361,0.0002866134,0.000046357392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00093328004,0.00079132756,0.0010434525,0.0017846938,0.0007315679,0.0007136018,0.0013241564,0.0011580228,0.0012165282],"category_scores_gemma":[0.002995925,0.00037461755,0.00070362474,0.0017856234,0.0004940955,0.00091837475,0.00088831055,0.00080995995,0.00035594316],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015849498,0.00016866972,0.0019216221,0.00013969428,0.00011121882,0.00017489503,0.0003335381,0.3359253,0.019105414,0.008948155,0.0031210703,0.6298919],"study_design_scores_gemma":[0.000029734143,0.00004689164,0.00037258657,0.000010022768,0.000019496669,0.00008781038,0.000035972276,0.99116194,0.0026493147,0.0033233152,0.0022502227,0.000012660098],"about_ca_topic_score_codex":0.003850426,"about_ca_topic_score_gemma":0.004103297,"teacher_disagreement_score":0.003850426,"about_ca_system_score_codex":0.000887514,"about_ca_system_score_gemma":0.001132459,"threshold_uncertainty_score":0.007656038},"labels":[],"label_agreement":null},{"id":"W2086368750","doi":"10.1145/371920.372162","title":"When experts agree","year":2001,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Watson; Citation; Library science; Research center; George (robot); Center (category theory); Computer science; World Wide Web; Operations research; Engineering; Political science; Artificial intelligence","score_opus":0.018819512951789196,"score_gpt":0.2764848424940115,"score_spread":0.2576653295422223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2086368750","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002605378,0.0016359313,0.007721998,0.47432557,0.1761157,0.002091706,0.004296521,0.0038616748,0.3273455],"genre_scores_gemma":[0.02824222,0.002163771,0.011237076,0.24308783,0.054591283,0.004868158,0.0046252,0.004096953,0.64708745],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.93595606,0.0135808885,0.0065677236,0.0047907755,0.030105457,0.008999176],"domain_scores_gemma":[0.74519837,0.034108464,0.008736487,0.0221131,0.15832645,0.031517036],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.036071967,0.0013457162,0.0017368408,0.00511602,0.007934975,0.02195543,0.004636337,0.025686814,0.41023543],"category_scores_gemma":[0.2671683,0.001192534,0.0020293023,0.0033699549,0.0032033478,0.019463364,0.01641389,0.015203821,0.34779534],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002556111,0.0000123510235,0.0001337129,0.00008035214,0.0000061571145,0.000068486734,0.00027010855,0.00001768661,0.000104499544,0.001195946,0.98367774,0.014407424],"study_design_scores_gemma":[0.000025153033,0.000014015281,0.00026413615,0.0002570184,0.000008688387,0.00005374502,0.001981174,0.000073850235,0.000191658,0.003989807,0.9930958,0.000044879016],"about_ca_topic_score_codex":0.0023121436,"about_ca_topic_score_gemma":0.0042036627,"teacher_disagreement_score":0.41023543,"about_ca_system_score_codex":0.0047267876,"about_ca_system_score_gemma":0.017155478,"threshold_uncertainty_score":0.8412276},"labels":[],"label_agreement":null},{"id":"W2087006618","doi":"10.1002/meet.2008.1450450206","title":"Concept theory and the role of conceptual coherence in assessments of similarity","year":2008,"lang":"en","type":"article","venue":"Proceedings of the American Society for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Coherence (philosophical gambling strategy); Similarity (geometry); Representation (politics); Computer science; Epistemology; Concept learning; Conceptual framework; Cognitive psychology; Cognitive science; Psychology; Artificial intelligence; Mathematics","score_opus":0.009150753586632495,"score_gpt":0.27932940321944766,"score_spread":0.2701786496328152,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2087006618","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37783667,0.008665043,0.48353082,0.011158015,0.00043204686,0.00039279158,0.0003839221,0.00025161973,0.117349125],"genre_scores_gemma":[0.97298765,0.00032723375,0.025815323,0.00021794248,0.00008350189,0.00014252767,0.00008296331,0.000024085426,0.00031868758],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97474957,0.016502067,0.0013002461,0.0022417095,0.0046780654,0.0005283772],"domain_scores_gemma":[0.8232033,0.1377481,0.015412666,0.0077269897,0.012946846,0.0029621385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02462101,0.0005936773,0.0009564128,0.012436991,0.0024860378,0.007846506,0.0018404925,0.0019899928,0.0045335316],"category_scores_gemma":[0.16523555,0.0005252225,0.00092705234,0.007481356,0.018398447,0.025764924,0.0077843363,0.003122495,0.00026660925],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016057433,0.00007369178,0.011019029,0.00036093584,0.0001800391,0.00014572505,0.011747792,0.0039517656,0.00057830306,0.9265182,0.0011704157,0.044093587],"study_design_scores_gemma":[0.000028872078,0.00006319794,0.0046384423,0.000108677836,0.000032457137,0.00008921803,0.0021622097,0.0089764325,0.00016841039,0.98173183,0.0019627258,0.000037478836],"about_ca_topic_score_codex":0.0024974048,"about_ca_topic_score_gemma":0.0016307925,"teacher_disagreement_score":0.02462101,"about_ca_system_score_codex":0.003868498,"about_ca_system_score_gemma":0.0018039002,"threshold_uncertainty_score":0.13020992},"labels":[],"label_agreement":null},{"id":"W2089286873","doi":"10.1037/h0087425","title":"The p-value fallacy and how to avoid it.","year":2003,"lang":"en","type":"article","venue":"Canadian Journal of Experimental Psychology/Revue canadienne de psychologie expérimentale","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Fallacy; Null hypothesis; Psychology; Value (mathematics); Statistical hypothesis testing; Interpretation (philosophy); p-value; Econometrics; Statistics; Alternative hypothesis; Inference; Cognitive psychology; Mathematics; Epistemology; Artificial intelligence; Computer science","score_opus":0.0283444507547641,"score_gpt":0.3147546726853978,"score_spread":0.2864102219306337,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2089286873","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018961844,0.03488715,0.78373647,0.14756456,0.015440333,0.00052827084,0.00034684272,0.0019447553,0.013655483],"genre_scores_gemma":[0.062139913,0.018147869,0.8131175,0.07936949,0.015273661,0.0037250135,0.0002446592,0.0016950256,0.0062868325],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.78726333,0.15454865,0.011962851,0.012138748,0.03325267,0.00083376287],"domain_scores_gemma":[0.4606815,0.48912403,0.012196805,0.01869696,0.017584784,0.0017159168],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.15100309,0.0032631317,0.0034606154,0.007539434,0.0033288638,0.0077407775,0.007008536,0.014447825,0.004870548],"category_scores_gemma":[0.5519866,0.002130364,0.0017987099,0.0075495634,0.037075832,0.018405171,0.007878626,0.026044156,0.0058773314],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031019,0.00012520171,0.002966444,0.0026853683,0.0006386502,0.0012323038,0.0026494183,0.0020125394,0.0008420717,0.531331,0.198807,0.25639978],"study_design_scores_gemma":[0.000095790434,0.000064776556,0.00053957227,0.001206663,0.00007654163,0.0010098189,0.00039107923,0.0037013663,0.0008594501,0.90679175,0.085128464,0.00013471219],"about_ca_topic_score_codex":0.0021817393,"about_ca_topic_score_gemma":0.0019640883,"teacher_disagreement_score":0.8489969,"about_ca_system_score_codex":0.003512912,"about_ca_system_score_gemma":0.005748644,"threshold_uncertainty_score":0.79859024},"labels":[],"label_agreement":null},{"id":"W2089787183","doi":"10.1109/cse.2014.69","title":"Automatic Twitter Topic Summarization","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University","funders":"","keywords":"Automatic summarization; Computer science; Ranking (information retrieval); Hidden Markov model; Multi-document summarization; Information retrieval; Social media; Bayesian probability; Parametric statistics; Artificial intelligence; Data mining; World Wide Web; Mathematics","score_opus":0.009246781099650433,"score_gpt":0.25368207453427477,"score_spread":0.24443529343462433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2089787183","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09876165,0.00687929,0.7800479,0.002222856,0.0018726556,0.0016314086,0.052666865,0.041252546,0.014664769],"genre_scores_gemma":[0.31357396,0.003826857,0.5002049,0.00028736758,0.0028443765,0.0016913443,0.15538613,0.00219845,0.019986577],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99890983,0.00021238958,0.00010496524,0.00028238716,0.00035245751,0.00013805069],"domain_scores_gemma":[0.99733007,0.00086084736,0.00031944973,0.00025150343,0.0011191408,0.00011897812],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012479257,0.0019284401,0.0013026958,0.007953992,0.000955223,0.0017486064,0.0011112574,0.0008980608,0.005957894],"category_scores_gemma":[0.0049857176,0.0004647242,0.0010028301,0.0045894063,0.0002339533,0.0020899782,0.001341001,0.0009868597,0.008056673],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011530401,0.00018309122,0.005190133,0.0015067469,0.00020840434,0.0005690762,0.00081705966,0.010642872,0.059198275,0.0039334116,0.12768151,0.78891647],"study_design_scores_gemma":[0.00024308516,0.0006435826,0.025132397,0.00035679585,0.00070881506,0.0012632549,0.002393504,0.60565215,0.11314854,0.024071213,0.22617692,0.00020973485],"about_ca_topic_score_codex":0.0016286364,"about_ca_topic_score_gemma":0.0025707409,"teacher_disagreement_score":0.007953992,"about_ca_system_score_codex":0.0004404278,"about_ca_system_score_gemma":0.00087709125,"threshold_uncertainty_score":0.019931138},"labels":[],"label_agreement":null},{"id":"W2091303901","doi":"10.1016/j.fss.2012.03.011","title":"Fuzzy logic and semiotic methods in modeling of medical concepts","year":2012,"lang":"en","type":"article","venue":"Fuzzy Sets and Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thompson Rivers University","funders":"","keywords":"Computer science; Representation (politics); Context (archaeology); Semiotics; Knowledge representation and reasoning; Ambiguity; Field (mathematics); Process (computing); Artificial intelligence; Data science; Theoretical computer science; Management science; Cognitive science; Epistemology; Mathematics; Programming language; Psychology","score_opus":0.06177873447235941,"score_gpt":0.4079266917271283,"score_spread":0.3461479572547689,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2091303901","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003607472,0.0065407776,0.9749799,0.0017061507,0.00030623574,0.00007332994,0.00015215503,0.00011930831,0.012514709],"genre_scores_gemma":[0.3279936,0.013483961,0.64491194,0.0006494839,0.0013178472,0.00056922575,0.00035964,0.000098052995,0.010616328],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976927,0.0011578084,0.00022168027,0.00019351678,0.00065953855,0.00007480043],"domain_scores_gemma":[0.9967733,0.0023586578,0.00020529567,0.00025782405,0.00032597268,0.00007896336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035311212,0.0008271707,0.0012092297,0.0034777995,0.0009336224,0.004297598,0.0014722386,0.0012789359,0.0036049183],"category_scores_gemma":[0.0064376416,0.0003903623,0.001622705,0.0025900886,0.0045252,0.004679143,0.0014923292,0.0019451446,0.00066956994],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015214795,0.000019692561,0.00010573414,0.0002141418,0.000028657372,0.000071021015,0.00033233134,0.013078733,0.0004793222,0.967134,0.0009904634,0.01753057],"study_design_scores_gemma":[0.0000103314405,0.000024862764,0.00009634073,0.000101405014,0.000023017818,0.000093266455,0.00014525016,0.04696637,0.0003957801,0.9382503,0.01387333,0.000019716068],"about_ca_topic_score_codex":0.0024471052,"about_ca_topic_score_gemma":0.0016133876,"teacher_disagreement_score":0.004297598,"about_ca_system_score_codex":0.0019936592,"about_ca_system_score_gemma":0.001772563,"threshold_uncertainty_score":0.018674612},"labels":[],"label_agreement":null},{"id":"W2091597866","doi":"10.1075/ssol.1.1.06dix","title":"The scientific study of literature","year":2011,"lang":"en","type":"article","venue":"Scientific Study of Literature","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reading (process); Context (archaeology); Cognition; Computer science; Cognitive science; Scientific literature; Domain (mathematical analysis); Event (particle physics); Psychology; Epistemology; Data science; Cognitive psychology; Neuroscience; Linguistics; History","score_opus":0.02404426383237939,"score_gpt":0.2786062161423486,"score_spread":0.2545619523099692,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2091597866","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0061988425,0.60892,0.05547628,0.16409408,0.087377116,0.00025815584,0.0004543987,0.00029946968,0.07692164],"genre_scores_gemma":[0.14885156,0.50040764,0.1038025,0.025201133,0.19408652,0.00077080535,0.0007634595,0.0003865738,0.025729902],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97130394,0.016218344,0.0026045346,0.001683436,0.007838313,0.00035142965],"domain_scores_gemma":[0.8285304,0.14033422,0.0075793047,0.0070304116,0.014311569,0.0022141726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022276722,0.00061646494,0.0011818903,0.018507795,0.0037068229,0.01284808,0.0017506384,0.0028863472,0.005049037],"category_scores_gemma":[0.06882325,0.00037350194,0.000890119,0.010561252,0.017560266,0.012831212,0.0030530263,0.00460439,0.0017756401],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043478438,0.00003355936,0.0012175418,0.0057320814,0.000096914795,0.0003484151,0.0113569945,0.00032466862,0.0007116737,0.5521671,0.13175847,0.296209],"study_design_scores_gemma":[0.000009752346,0.00003778428,0.0013945232,0.004731976,0.000030007726,0.0005556417,0.003776447,0.00033319465,0.00030898524,0.2482782,0.74051034,0.000033087104],"about_ca_topic_score_codex":0.0008603008,"about_ca_topic_score_gemma":0.0011767811,"teacher_disagreement_score":0.022276722,"about_ca_system_score_codex":0.0029472765,"about_ca_system_score_gemma":0.0057521174,"threshold_uncertainty_score":0.11781192},"labels":[],"label_agreement":null},{"id":"W2091855347","doi":"10.5539/cis.v7n4p123","title":"Ontology Based Data Mining Approach on Web Documents","year":2014,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Ontology; Information retrieval; Web mining; The Internet; World Wide Web; Data Web; Data mining; Web page","score_opus":0.02619256338547216,"score_gpt":0.2971963681470349,"score_spread":0.2710038047615627,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2091855347","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10686019,0.0023383833,0.8676284,0.0018002472,0.0001765287,0.0013990676,0.0048606033,0.0023000618,0.012636534],"genre_scores_gemma":[0.20668963,0.0020707746,0.7802024,0.00022401474,0.00006211446,0.0006375974,0.0057426416,0.00007842948,0.0042924467],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981008,0.0002662325,0.0002969036,0.0002937024,0.0009483664,0.00009396751],"domain_scores_gemma":[0.9988626,0.00044615564,0.00013541894,0.000112190144,0.00037920987,0.00006445463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010251685,0.00043691337,0.0007900383,0.006296629,0.0014187009,0.0027387375,0.0010888812,0.00071397563,0.0011274489],"category_scores_gemma":[0.003199229,0.00021703354,0.0014443714,0.006355691,0.000602934,0.002167232,0.0009856771,0.0009960913,0.00054039364],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026975104,0.00083374284,0.012873534,0.0013558936,0.00029306757,0.0017288178,0.002099208,0.027333923,0.025366971,0.04989486,0.010621345,0.86732894],"study_design_scores_gemma":[0.00010046937,0.00025936673,0.017152335,0.000647439,0.00038575972,0.0028718524,0.00443804,0.6448218,0.062268224,0.09350768,0.17336242,0.00018465405],"about_ca_topic_score_codex":0.007342746,"about_ca_topic_score_gemma":0.008985581,"teacher_disagreement_score":0.007342746,"about_ca_system_score_codex":0.0013477332,"about_ca_system_score_gemma":0.002658244,"threshold_uncertainty_score":0.0146000385},"labels":[],"label_agreement":null},{"id":"W2094256961","doi":"10.1145/2661829.2661963","title":"Succinct Queries for Linking and Tracking News in Social Media","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Information retrieval; Social media; Set (abstract data type); Rank (graph theory); Key (lock); World Wide Web; Mathematics","score_opus":0.023657716101981644,"score_gpt":0.29482792920784595,"score_spread":0.2711702131058643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2094256961","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07513975,0.0012619587,0.8830043,0.0007642868,0.00015305122,0.0018857131,0.01506026,0.01819788,0.004532808],"genre_scores_gemma":[0.17109242,0.00049594376,0.80369806,0.00028911902,0.00014976882,0.0011582039,0.01900323,0.00091875123,0.0031945503],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9950761,0.0014053572,0.0006706275,0.0007157908,0.0019116594,0.00022043001],"domain_scores_gemma":[0.9854879,0.008669981,0.00134339,0.002175017,0.0019307401,0.00039296353],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003071161,0.0015881592,0.0009298371,0.0063720513,0.0010799026,0.0025680498,0.0015213208,0.0020256562,0.0048043258],"category_scores_gemma":[0.024537167,0.0006780987,0.0008319629,0.005013587,0.00092096446,0.00806161,0.002717596,0.0014270269,0.003019528],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002242152,0.0010892256,0.0163438,0.0023240205,0.00020405723,0.0014918124,0.006700175,0.030254727,0.119107135,0.03281982,0.060828067,0.726595],"study_design_scores_gemma":[0.00027406967,0.0011258061,0.010065119,0.00026466843,0.00018946033,0.0025446154,0.0038925877,0.7295472,0.09537485,0.05038824,0.106021,0.00031232828],"about_ca_topic_score_codex":0.004022123,"about_ca_topic_score_gemma":0.006723511,"teacher_disagreement_score":0.0063720513,"about_ca_system_score_codex":0.00090289046,"about_ca_system_score_gemma":0.00092256867,"threshold_uncertainty_score":0.016242027},"labels":[],"label_agreement":null},{"id":"W2096786223","doi":"10.1109/nafips.2004.1337433","title":"On the direct scaling approach of eliciting aggregated fuzzy information: the psychophysical view","year":2004,"lang":"en","type":"article","venue":"IEEE Annual Meeting of the Fuzzy Information, 2004. Processing NAFIPS '04.","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University","funders":"","keywords":"Scaling; Fuzzy logic; Operator (biology); Simple (philosophy); Computer science; Fuzzy set; Artificial intelligence; Mathematics","score_opus":0.010723362078098337,"score_gpt":0.2514054052676692,"score_spread":0.24068204318957084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096786223","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09538493,0.0015892694,0.8244327,0.003637617,0.0004109974,0.0004869627,0.00015330662,0.00060278905,0.07330145],"genre_scores_gemma":[0.72905695,0.0010798221,0.26344255,0.0014126034,0.00024136218,0.0008201639,0.00007031553,0.00013333699,0.0037429873],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9967765,0.0013779803,0.00013497911,0.0006186012,0.0010180954,0.00007383757],"domain_scores_gemma":[0.9893561,0.006969036,0.0007229235,0.0019489973,0.0008007782,0.00020208475],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003670526,0.0009040931,0.00040276753,0.0009786782,0.00054827274,0.0020659785,0.0010889356,0.0012888365,0.006613118],"category_scores_gemma":[0.018706108,0.0003799664,0.0007024729,0.00043401797,0.003981192,0.0049910354,0.0019341451,0.0017226549,0.0007376767],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008883772,0.00052657875,0.0028274697,0.001596809,0.00015115336,0.0004043477,0.0032184434,0.009866813,0.20612644,0.53668493,0.003040099,0.23466845],"study_design_scores_gemma":[0.00023896428,0.0016059277,0.0101367645,0.00033234255,0.00013835957,0.0015329833,0.0011178204,0.079240106,0.08140557,0.80368567,0.020349458,0.00021603245],"about_ca_topic_score_codex":0.00034900432,"about_ca_topic_score_gemma":0.0002556218,"teacher_disagreement_score":0.006613118,"about_ca_system_score_codex":0.00059642486,"about_ca_system_score_gemma":0.00045487427,"threshold_uncertainty_score":0.022123039},"labels":[],"label_agreement":null},{"id":"W2097212070","doi":"10.3389/fpsyg.2015.01447","title":"Quantum structure of negation and conjunction in human thought","year":2015,"lang":"en","type":"article","venue":"Frontiers in Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; Kelowna General Hospital","funders":"","keywords":"Conjunction (astronomy); Negation; Representation (politics); Set (abstract data type); Space (punctuation); Fuzzy set; Mathematics; Fuzzy logic; Computer science; Artificial intelligence; Theoretical computer science; Algebra over a field; Pure mathematics","score_opus":0.018490580233547427,"score_gpt":0.32598122085839437,"score_spread":0.30749064062484693,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2097212070","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8146259,0.00057427556,0.16292268,0.0013426996,0.000056567078,0.00006199066,0.00014463133,0.00016160507,0.020109631],"genre_scores_gemma":[0.9872056,0.000059610047,0.0123369275,0.00007075487,0.000020362035,0.000028481882,0.00003291967,0.000012710583,0.00023258977],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9952212,0.0026406709,0.00014885652,0.000648309,0.001164009,0.00017690033],"domain_scores_gemma":[0.9805685,0.015357523,0.0017125873,0.0014118939,0.00072282687,0.0002265658],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003888569,0.00024226704,0.0004805862,0.0012795627,0.00097192003,0.002189766,0.0005185575,0.0007621101,0.003176914],"category_scores_gemma":[0.029768266,0.00033378787,0.00042413294,0.0012209249,0.0065886634,0.0048521603,0.0016268936,0.0010522316,0.00018317104],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013864621,0.00022232484,0.027071413,0.0007283295,0.0002259201,0.000520401,0.0248053,0.020426085,0.035228495,0.77622736,0.0012527946,0.11190515],"study_design_scores_gemma":[0.00008603414,0.00019515945,0.033495706,0.000047715537,0.000057080037,0.00035564962,0.0022627064,0.047732163,0.0042363866,0.90781134,0.0036111006,0.00010903972],"about_ca_topic_score_codex":0.0011131269,"about_ca_topic_score_gemma":0.0006320185,"teacher_disagreement_score":0.003888569,"about_ca_system_score_codex":0.0011036162,"about_ca_system_score_gemma":0.00048710365,"threshold_uncertainty_score":0.020564914},"labels":[],"label_agreement":null},{"id":"W2100784475","doi":"10.1177/1461445612466468","title":"Rhetorical relations in multimodal documents","year":2013,"lang":"en","type":"article","venue":"Discourse Studies","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Rhetorical question; Presentational and representational acting; Computer science; Natural language processing; Coherence (philosophical gambling strategy); Linguistics; Categorization; Set (abstract data type); Artificial intelligence; Subject (documents); Mathematics; World Wide Web; Philosophy","score_opus":0.0271779675503826,"score_gpt":0.37800327637377823,"score_spread":0.35082530882339563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100784475","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91692924,0.004673783,0.0495817,0.001582393,0.00008736658,0.00018403833,0.002147302,0.00026525793,0.024548935],"genre_scores_gemma":[0.9757976,0.0006843722,0.020392016,0.00006422755,0.000060533308,0.00018763408,0.0009142943,0.00008585395,0.0018135426],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9968194,0.0018504013,0.00019788132,0.00036439256,0.00065068505,0.00011720716],"domain_scores_gemma":[0.96462506,0.029300224,0.0025674156,0.001253811,0.0019566116,0.00029686082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030575315,0.00029324618,0.0004401443,0.008957287,0.0033281993,0.004170448,0.0005977471,0.00087099714,0.0039281934],"category_scores_gemma":[0.026955487,0.00024701445,0.00022163837,0.010255419,0.0034545527,0.0057202224,0.0019566156,0.0012984229,0.00038172308],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059044815,0.00039219434,0.040193714,0.0033579108,0.00011474711,0.0035198643,0.39090705,0.006334382,0.028176619,0.24852474,0.00919554,0.2686928],"study_design_scores_gemma":[0.00018321475,0.00044556631,0.16458538,0.0018493307,0.00028774343,0.0038708462,0.19842754,0.043630816,0.037467796,0.16815569,0.38072678,0.00036929973],"about_ca_topic_score_codex":0.0031781646,"about_ca_topic_score_gemma":0.0047039143,"teacher_disagreement_score":0.008957287,"about_ca_system_score_codex":0.0019685547,"about_ca_system_score_gemma":0.00094561186,"threshold_uncertainty_score":0.016169906},"labels":[],"label_agreement":null},{"id":"W2101145506","doi":"10.1002/meet.14504701412","title":"How hierarchical structures may influence the way that we think","year":2010,"lang":"en","type":"article","venue":"Proceedings of the American Society for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Hierarchy; Computer science; Task (project management); Hierarchical database model; Affect (linguistics); Hierarchical organization; Cognitive psychology; Psychology; Data mining; Communication; Engineering; Systems engineering; Political science","score_opus":0.008299225593631067,"score_gpt":0.261755658319421,"score_spread":0.25345643272578994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2101145506","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9312435,0.00017623018,0.044374235,0.0009077108,0.00006037311,0.000097665434,0.00012040304,0.000510817,0.022509113],"genre_scores_gemma":[0.9881557,0.000050856623,0.010554041,0.00010225083,0.000011228713,0.000030163896,0.000069702946,0.000068144145,0.0009579108],"study_design_codex":"bench_or_experimental","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9979165,0.0011214927,0.00007167242,0.00028892083,0.00046370013,0.0001377588],"domain_scores_gemma":[0.9798696,0.013640711,0.0019735747,0.0018106394,0.0018329072,0.0008725027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023789124,0.0002820236,0.00014274416,0.00056744873,0.0006658394,0.003154285,0.00043471772,0.0006876939,0.0048632477],"category_scores_gemma":[0.026075648,0.00038249677,0.00023965785,0.0003385036,0.001346774,0.0037743337,0.0007122593,0.001003676,0.00071340974],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030729382,0.0009535596,0.14530198,0.00087961706,0.00035188493,0.0012640324,0.16524366,0.009174037,0.2848774,0.115856804,0.017480478,0.25554353],"study_design_scores_gemma":[0.00081186666,0.0022589425,0.46167547,0.00050258357,0.0006245929,0.001546163,0.041639764,0.12157569,0.07215324,0.22529864,0.07133875,0.0005742504],"about_ca_topic_score_codex":0.0019804114,"about_ca_topic_score_gemma":0.0022249646,"teacher_disagreement_score":0.0048632477,"about_ca_system_score_codex":0.00057106686,"about_ca_system_score_gemma":0.00032534433,"threshold_uncertainty_score":0.016269207},"labels":[],"label_agreement":null},{"id":"W2102724816","doi":"10.1109/kam.2009.191","title":"PH-SSBM: Phrase Semantic Similarity Based Model for Document Clustering","year":2009,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"WordNet; Semantic similarity; Computer science; Natural language processing; Artificial intelligence; Document clustering; Explicit semantic analysis; Cluster analysis; Similarity (geometry); Phrase; Semantic computing; Representation (politics); tf–idf; Information retrieval; Semantic technology; Semantic Web; Term (time)","score_opus":0.024105504912120004,"score_gpt":0.3109896130239714,"score_spread":0.2868841081118514,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102724816","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0050465944,0.0004761916,0.9913851,0.00018700308,0.000065745684,0.00015371836,0.0004645008,0.0011355795,0.0010855652],"genre_scores_gemma":[0.16590032,0.0009907944,0.8229842,0.0003025781,0.00019477049,0.0010157725,0.0038683093,0.00032595283,0.004417375],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973212,0.00095212605,0.00020809635,0.0004811568,0.0009473623,0.00009016663],"domain_scores_gemma":[0.9985469,0.0005409203,0.00012098801,0.00027345194,0.0004654403,0.000052338743],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020041256,0.0009455816,0.0014302582,0.0035905065,0.0008995489,0.0016371025,0.0029420613,0.0015548503,0.002511546],"category_scores_gemma":[0.0055484595,0.000393215,0.0015935419,0.004765775,0.00071989995,0.004476046,0.0013919459,0.0011839365,0.0021938996],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006000937,0.00032074327,0.0029884346,0.0007843461,0.00046073546,0.00023931483,0.0007250935,0.18438026,0.012507892,0.080549344,0.016967146,0.69947666],"study_design_scores_gemma":[0.000041923853,0.00013679033,0.0008216967,0.000048661565,0.000082656756,0.00022652197,0.00010565107,0.9228024,0.004411374,0.058786005,0.012482919,0.000053350777],"about_ca_topic_score_codex":0.005384274,"about_ca_topic_score_gemma":0.006015493,"teacher_disagreement_score":0.005384274,"about_ca_system_score_codex":0.0015444623,"about_ca_system_score_gemma":0.0016814467,"threshold_uncertainty_score":0.011205912},"labels":[],"label_agreement":null},{"id":"W2106377293","doi":"10.1109/hicss.1999.772650","title":"The functionality attribute of cybergenres","year":2003,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":106,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Tuple; The Internet; Class (philosophy); Content (measure theory); World Wide Web; Multimedia; Information retrieval; Artificial intelligence; Mathematics","score_opus":0.015531196943938437,"score_gpt":0.26695244195383044,"score_spread":0.251421245009892,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106377293","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55450153,0.0014062152,0.077486835,0.0022270277,0.00056080957,0.00020734231,0.0011905173,0.0012411304,0.3611786],"genre_scores_gemma":[0.967967,0.00046121454,0.013725175,0.00018652141,0.00026721627,0.000066503526,0.00076604943,0.0002175401,0.016342817],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9975048,0.0007095598,0.00033358045,0.0002624743,0.0009901426,0.00019944926],"domain_scores_gemma":[0.98894686,0.0042402525,0.0014990743,0.0019974024,0.0024864452,0.00082996074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001608356,0.00047207263,0.00022746676,0.004067788,0.001732007,0.0043690642,0.0005110772,0.00084813376,0.0051026694],"category_scores_gemma":[0.0120509975,0.00019331193,0.0004121122,0.002163846,0.002904566,0.006150357,0.0018143402,0.0009514745,0.0009889402],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032667926,0.000117646494,0.047787014,0.00066181395,0.000044426703,0.001999137,0.040654887,0.0012941405,0.026007907,0.6084034,0.009414103,0.26328883],"study_design_scores_gemma":[0.000046544625,0.00042219314,0.13951984,0.00068914995,0.00012707087,0.013946167,0.025867807,0.01195535,0.017017432,0.24704032,0.5431498,0.00021833289],"about_ca_topic_score_codex":0.0012908193,"about_ca_topic_score_gemma":0.0011102706,"teacher_disagreement_score":0.0051026694,"about_ca_system_score_codex":0.001131849,"about_ca_system_score_gemma":0.00055551756,"threshold_uncertainty_score":0.017070115},"labels":[],"label_agreement":null},{"id":"W2106900819","doi":"10.1109/hicss.2002.994040","title":"Adaptive user modeling for filtering electronic news","year":2003,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Newspaper; Artificial neural network; User modeling; Task (project management); Quality (philosophy); User profile; Adaptive system; Test (biology); Information retrieval; Multimedia; Artificial intelligence; World Wide Web; User interface; Advertising; Engineering","score_opus":0.026812736624997772,"score_gpt":0.28011909598136164,"score_spread":0.25330635935636386,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106900819","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1487294,0.00015630874,0.8404304,0.00026353705,0.000041902476,0.00041476593,0.0002616361,0.007321619,0.0023805206],"genre_scores_gemma":[0.7363641,0.00012758636,0.25780714,0.00016923816,0.000029726749,0.0004803162,0.0005596742,0.00015149868,0.0043107052],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.998428,0.00083923,0.00009626093,0.00028251024,0.0002720681,0.00008195692],"domain_scores_gemma":[0.9940499,0.0041673346,0.0002694097,0.0005066422,0.00086332875,0.00014356518],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00295561,0.0006917208,0.0007912363,0.0006985845,0.00037348946,0.000998821,0.0010755633,0.0010136947,0.0027544193],"category_scores_gemma":[0.010581196,0.00040281494,0.00057514815,0.00042138025,0.00032330022,0.0013347828,0.0006165451,0.00094316347,0.0013378162],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003445451,0.0016912585,0.026091823,0.0005280719,0.00031634726,0.00059046235,0.004503584,0.17331704,0.055666547,0.007390489,0.00919453,0.7172645],"study_design_scores_gemma":[0.00004558478,0.00024923138,0.0026628056,0.000015098643,0.000041964162,0.00013925284,0.00014285411,0.9850344,0.0070256214,0.0023287938,0.0022780136,0.00003640857],"about_ca_topic_score_codex":0.0063581364,"about_ca_topic_score_gemma":0.0070497026,"teacher_disagreement_score":0.0063581364,"about_ca_system_score_codex":0.0006611257,"about_ca_system_score_gemma":0.0005121036,"threshold_uncertainty_score":0.01563096},"labels":[],"label_agreement":null},{"id":"W2107363066","doi":"10.1002/asi.23367","title":"The invariant distribution of references in scientific articles","year":2015,"lang":"en","type":"article","venue":"Journal of the Association for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"Canada Research Chairs","keywords":"Bibliometrics; Computer science; Scientific communication; Section (typography); Information retrieval; Library science; Data science","score_opus":0.01949208080998927,"score_gpt":0.27803940768089846,"score_spread":0.2585473268709092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2107363066","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9504974,0.0042828256,0.019091219,0.0009115088,0.00019667263,0.000075320124,0.0022369418,0.00032495038,0.022383079],"genre_scores_gemma":[0.99397284,0.0007696851,0.0024605226,0.00007877742,0.00019118602,0.000040831084,0.0008817585,0.00007744744,0.0015269209],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9898689,0.0025080007,0.0016692808,0.002468429,0.0028974311,0.00058801036],"domain_scores_gemma":[0.8588831,0.073624186,0.03389973,0.014555574,0.017354853,0.0016824811],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.009893283,0.00019662891,0.0006025152,0.014625294,0.00096554804,0.00517391,0.0008142925,0.0008326152,0.0049345796],"category_scores_gemma":[0.12550648,0.00029892786,0.0003894501,0.018021543,0.0029073586,0.0061279805,0.0023911837,0.0007844868,0.002033937],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007818285,0.00012984991,0.68601876,0.0007601377,0.0004714213,0.0007531086,0.015192258,0.0025036617,0.012139184,0.0603211,0.004328539,0.21660021],"study_design_scores_gemma":[0.000045762525,0.00026568893,0.8891271,0.00024904808,0.00023531685,0.0011673556,0.006177196,0.0048021763,0.0053915726,0.07247133,0.019923942,0.00014337886],"about_ca_topic_score_codex":0.0014666762,"about_ca_topic_score_gemma":0.0010930778,"teacher_disagreement_score":0.9853747,"about_ca_system_score_codex":0.001327081,"about_ca_system_score_gemma":0.00081819674,"threshold_uncertainty_score":0.052321315},"labels":[],"label_agreement":null},{"id":"W2109753595","doi":"10.1109/lsp.2010.2048940","title":"Self-Organizing Maps for Topic Trend Discovery","year":2010,"lang":"en","type":"article","venue":"IEEE Signal Processing Letters","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"Mitacs","keywords":"Latent Dirichlet allocation; Computer science; Topic model; Visualization; Dimensionality reduction; Data mining; Information retrieval; Set (abstract data type); Data visualization; Process (computing); Latent semantic analysis; Curse of dimensionality; Artificial intelligence; Machine learning","score_opus":0.01012035866637828,"score_gpt":0.25203025302561993,"score_spread":0.24190989435924165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109753595","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018854033,0.0012495394,0.97082,0.00041879143,0.00013655052,0.0001406884,0.0013974893,0.0042327163,0.0027502063],"genre_scores_gemma":[0.28762653,0.001143065,0.70323855,0.00010747635,0.00031823808,0.00064026954,0.0031424295,0.0004510318,0.0033323576],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99897206,0.00039329167,0.00005543385,0.00019789323,0.00030355796,0.00007780097],"domain_scores_gemma":[0.9968953,0.0018181613,0.00029109424,0.00040193845,0.00051224796,0.000081340004],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017848595,0.00091269845,0.00091753027,0.0058499384,0.0009168875,0.0019006929,0.0012494511,0.0009607089,0.0028993424],"category_scores_gemma":[0.008497552,0.0005529106,0.0011475579,0.0050193043,0.00048349774,0.0018477328,0.0012153103,0.0010304811,0.0013256972],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033558873,0.00020007334,0.0057226443,0.0005551165,0.0003877735,0.0004312099,0.0010263415,0.21231395,0.00548469,0.05372649,0.030754024,0.68906206],"study_design_scores_gemma":[0.000016667673,0.000021043767,0.0014671191,0.000025115742,0.00002418199,0.0001013023,0.00015657241,0.9414755,0.0019387491,0.046638858,0.0080969455,0.00003792335],"about_ca_topic_score_codex":0.004405148,"about_ca_topic_score_gemma":0.004498892,"teacher_disagreement_score":0.0058499384,"about_ca_system_score_codex":0.0006842669,"about_ca_system_score_gemma":0.0009030539,"threshold_uncertainty_score":0.009699285},"labels":[],"label_agreement":null},{"id":"W2111310810","doi":"10.1145/1148170.1148262","title":"Statistical precision of information retrieval evaluation","year":2006,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Data mining; Concordance; Population; Information retrieval; Confidence interval; Statistics; Degree (music); Statistical hypothesis testing; Data collection; Test (biology); Artificial intelligence; Mathematics","score_opus":0.011818057569659177,"score_gpt":0.3080099634055263,"score_spread":0.2961919058358671,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111310810","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045771748,0.006599812,0.93357146,0.00075712905,0.0003260046,0.00026459168,0.0012230967,0.0029272398,0.008558953],"genre_scores_gemma":[0.82237023,0.0013934103,0.16920805,0.0006164747,0.00073985214,0.00089645234,0.0023613395,0.0013174659,0.0010967402],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.8319094,0.08507289,0.010914503,0.019586086,0.05027729,0.002239799],"domain_scores_gemma":[0.33632085,0.52526194,0.026998842,0.08046687,0.029973615,0.0009778818],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.13682042,0.0017324655,0.0033670848,0.012271015,0.0017754685,0.0069555435,0.004013717,0.0035727136,0.001975494],"category_scores_gemma":[0.5522661,0.0012935873,0.0025638286,0.011163847,0.0046697436,0.009290813,0.0048650457,0.004400136,0.001214192],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011970076,0.00025557465,0.09124732,0.0020213195,0.0029927315,0.00035644977,0.0021137497,0.2885578,0.0037589602,0.15622607,0.013729233,0.43754378],"study_design_scores_gemma":[0.00015353506,0.00065786886,0.04053127,0.00071536785,0.00067855837,0.00090747344,0.00032445922,0.6826802,0.012847733,0.2501026,0.010019581,0.00038144112],"about_ca_topic_score_codex":0.0038052641,"about_ca_topic_score_gemma":0.0029275948,"teacher_disagreement_score":0.13682042,"about_ca_system_score_codex":0.0035074279,"about_ca_system_score_gemma":0.002702758,"threshold_uncertainty_score":0.7235842},"labels":[],"label_agreement":null},{"id":"W2111823757","doi":"10.1109/icccyb.2010.5491333","title":"A new search method for ranking short text messages using semantic features and cluster coherence","year":2010,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Coherence (philosophical gambling strategy); Ranking (information retrieval); Information retrieval; Artificial intelligence; Semantic similarity; Measure (data warehouse); Natural language processing; Similarity (geometry); Non-negative matrix factorization; Cluster (spacecraft); Matrix decomposition; Data mining; Mathematics; Statistics; Image (mathematics)","score_opus":0.028520676570858695,"score_gpt":0.37075938864443203,"score_spread":0.3422387120735733,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111823757","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01135183,0.0006772668,0.9842173,0.00014193312,0.000084086656,0.0002388157,0.00038162494,0.0017549491,0.0011522366],"genre_scores_gemma":[0.11660133,0.0003562413,0.8771422,0.00008768961,0.00018269566,0.00035992207,0.001248696,0.00026362322,0.0037576125],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978441,0.00038608778,0.00017957336,0.00037269582,0.0011104277,0.000107029955],"domain_scores_gemma":[0.9974776,0.0009924115,0.00029202775,0.0002190721,0.0009057479,0.00011315302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014464319,0.0010843679,0.0016935421,0.007660247,0.0009610426,0.0017184576,0.0014263293,0.0010896474,0.0038404525],"category_scores_gemma":[0.0052239154,0.00044885074,0.0008793887,0.0054612714,0.0005149681,0.002743366,0.0009692361,0.00070101104,0.0015350956],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006063659,0.0003493595,0.003088266,0.00071614765,0.0003191535,0.0001693701,0.0003456605,0.016435001,0.02997055,0.013404758,0.01579511,0.9188003],"study_design_scores_gemma":[0.000280547,0.0006098136,0.0057579996,0.00007985211,0.0002361123,0.0008011944,0.0003751475,0.93212473,0.01951312,0.021221006,0.018777704,0.00022266874],"about_ca_topic_score_codex":0.005005459,"about_ca_topic_score_gemma":0.00905769,"teacher_disagreement_score":0.007660247,"about_ca_system_score_codex":0.0009029881,"about_ca_system_score_gemma":0.0014065445,"threshold_uncertainty_score":0.012847602},"labels":[],"label_agreement":null},{"id":"W2114224956","doi":"10.1080/15326900701326600","title":"What Makes People Revise Their Beliefs Following Contradictory Anecdotal Evidence?: The Role of Systemic Variability and Direct Experience","year":2007,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Randomness; Element (criminal law); Generality; Psychology; Certainty; Action (physics); Social psychology; Cognitive psychology; Mathematics; Statistics","score_opus":0.014261907045873835,"score_gpt":0.29217931190000923,"score_spread":0.2779174048541354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2114224956","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9951407,0.00030293985,0.0026715999,0.00024519343,0.000009858066,0.000023345381,0.00001547083,0.000015665099,0.0015753281],"genre_scores_gemma":[0.9988636,0.00012197735,0.0008395358,0.000064928754,0.00000786136,0.000011123742,0.00001698423,0.0000050536455,0.000068949084],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9884233,0.007498913,0.00063774677,0.0011666986,0.0018328744,0.00044060085],"domain_scores_gemma":[0.7218338,0.20875308,0.043634556,0.015489979,0.008146432,0.0021420692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013862312,0.00033302122,0.00046087778,0.0010854825,0.00044192033,0.0032645436,0.0006269039,0.0011781394,0.0009379149],"category_scores_gemma":[0.17805897,0.00063077797,0.00048308176,0.0006185354,0.0030223313,0.0029971146,0.0014101337,0.0013860664,0.00012262264],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002537687,0.00064347504,0.7020455,0.0011367551,0.0014982332,0.00115241,0.10572619,0.0029688536,0.021738498,0.0033676603,0.0005327046,0.1566521],"study_design_scores_gemma":[0.0001950961,0.001202955,0.94263184,0.00036506602,0.0005640573,0.0016643993,0.02485695,0.008175196,0.0060877856,0.01195821,0.0019837953,0.00031470208],"about_ca_topic_score_codex":0.0018688544,"about_ca_topic_score_gemma":0.0020416968,"teacher_disagreement_score":0.013862312,"about_ca_system_score_codex":0.0005799582,"about_ca_system_score_gemma":0.0005761473,"threshold_uncertainty_score":0.073311806},"labels":[],"label_agreement":null},{"id":"W2115756143","doi":"10.5430/ijhe.v2n4p172","title":"The Integrative Model of Behavior Prediction to Explain Technology Use in Post-graduate Teacher Education Programs in the Netherlands","year":2013,"lang":"en","type":"article","venue":"International Journal of Higher Education","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Post graduate; Graduate education; Mathematics education; Graduate students; Psychology; Pedagogy; Medical education; Medicine","score_opus":0.03703895631413239,"score_gpt":0.3435336310024244,"score_spread":0.306494674688292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115756143","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97842604,0.00043944785,0.012965505,0.00054067466,0.00002227186,0.00009258604,0.00013137987,0.000032400178,0.0073497156],"genre_scores_gemma":[0.9960802,0.00019104056,0.00257295,0.000025820398,0.0000039796682,0.00007774958,0.0001355369,0.000010586943,0.0009020632],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99845576,0.0008462101,0.000067662826,0.00019936012,0.00022521905,0.00020587786],"domain_scores_gemma":[0.99441,0.0040519424,0.00057570386,0.00021483246,0.00045417095,0.00029329356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002909163,0.00052218203,0.00051286444,0.0012014998,0.0005579282,0.0017896122,0.0010478831,0.00058789476,0.0036528788],"category_scores_gemma":[0.012566971,0.000487164,0.00086031144,0.0009569652,0.00074716983,0.0013705629,0.0011064848,0.0009884067,0.0003213817],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022760598,0.0013737954,0.8483226,0.0002535559,0.0003975251,0.00066755514,0.018146385,0.022042843,0.0007530977,0.024678882,0.0017932261,0.08134296],"study_design_scores_gemma":[0.00016217644,0.00035257818,0.71417665,0.0003223388,0.00029878237,0.0004526004,0.013916702,0.23103258,0.00068386167,0.031008644,0.0075082476,0.000084850035],"about_ca_topic_score_codex":0.05445561,"about_ca_topic_score_gemma":0.031915437,"teacher_disagreement_score":0.05445561,"about_ca_system_score_codex":0.002395461,"about_ca_system_score_gemma":0.0021290092,"threshold_uncertainty_score":0.10827726},"labels":[],"label_agreement":null},{"id":"W2116373577","doi":"10.1109/nafips.2004.1337356","title":"A fuzzy set approach to extracting keywords from abstracts","year":2004,"lang":"en","type":"article","venue":"IEEE Annual Meeting of the Fuzzy Information, 2004. Processing NAFIPS '04.","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Artificial intelligence; Fuzzy set; Fuzzy logic; Relevance (law); Set (abstract data type); Vocabulary; Fuzzy clustering; Natural language processing; Data mining; Fuzzy classification; Natural language; Information retrieval; Cluster analysis; Linguistics","score_opus":0.01571152251035344,"score_gpt":0.2666254911786536,"score_spread":0.2509139686683001,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2116373577","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0039918222,0.0010325866,0.99030626,0.00022000924,0.00012150817,0.0002850106,0.00031711286,0.00067063177,0.0030551276],"genre_scores_gemma":[0.026320847,0.0006905467,0.9689941,0.00007073204,0.00008686692,0.00040262102,0.0004640103,0.000049607137,0.0029207512],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99794966,0.00038822187,0.00032648782,0.0004357156,0.00082613743,0.00007389444],"domain_scores_gemma":[0.9983998,0.0005621753,0.00013292654,0.00011940871,0.0007273913,0.000058181395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015083482,0.0010095994,0.0012973264,0.005998476,0.0015294118,0.002321532,0.0018268296,0.0010212215,0.0024566874],"category_scores_gemma":[0.0039930935,0.00049738697,0.001669088,0.003965661,0.00088328356,0.0021010602,0.0008701949,0.0010992482,0.0012735467],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019948758,0.00017190265,0.0011496086,0.0013532599,0.00027810846,0.0005340156,0.0010802195,0.023278039,0.04624304,0.051971532,0.007516254,0.86622447],"study_design_scores_gemma":[0.00014994958,0.000609374,0.003930561,0.00058172416,0.0006152942,0.002414517,0.0014104702,0.57903814,0.08587299,0.18280981,0.14213896,0.00042817672],"about_ca_topic_score_codex":0.0042336434,"about_ca_topic_score_gemma":0.00534093,"teacher_disagreement_score":0.005998476,"about_ca_system_score_codex":0.0014992865,"about_ca_system_score_gemma":0.002748538,"threshold_uncertainty_score":0.010878086},"labels":[],"label_agreement":null},{"id":"W2117624436","doi":"","title":"Bike: Bilingual Keyphrase Experiments","year":2005,"lang":"en","type":"article","venue":"NPARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Machine translation; Information retrieval; Resource (disambiguation)","score_opus":0.01699604029499381,"score_gpt":0.30992152506400705,"score_spread":0.29292548476901326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2117624436","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8037132,0.0026964105,0.08260033,0.0016532937,0.0014147118,0.0023405426,0.023324858,0.021634368,0.060622204],"genre_scores_gemma":[0.8214262,0.0005204938,0.1187787,0.00091744255,0.00017690222,0.0022107393,0.03209884,0.0023319281,0.02153875],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969824,0.0016119897,0.0002785821,0.0005308376,0.00040796606,0.00018812635],"domain_scores_gemma":[0.9886193,0.0076321745,0.000296673,0.0017871396,0.001253053,0.00041169216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003826686,0.0013451012,0.0010723359,0.0010487459,0.0017747927,0.0012084909,0.0015074061,0.0020809418,0.019260453],"category_scores_gemma":[0.01555805,0.00045737697,0.0006046298,0.0013381591,0.00064143044,0.0029098198,0.0019685596,0.0020194387,0.007536045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.037921418,0.016393073,0.01023898,0.0077792834,0.001219912,0.002458859,0.0048793387,0.036808092,0.15006149,0.02546483,0.19973576,0.50703895],"study_design_scores_gemma":[0.012286319,0.019351255,0.026903508,0.0004999197,0.00078883563,0.003823682,0.0049257395,0.37278742,0.24663149,0.046742227,0.26454273,0.0007169448],"about_ca_topic_score_codex":0.003062127,"about_ca_topic_score_gemma":0.0031054753,"teacher_disagreement_score":0.019260453,"about_ca_system_score_codex":0.0006018698,"about_ca_system_score_gemma":0.00091695174,"threshold_uncertainty_score":0.06443268},"labels":[],"label_agreement":null},{"id":"W2118960523","doi":"10.1145/1242572.1242829","title":"Generating efficient labels to facilitate web accessibility","year":2007,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Web page; Set (abstract data type); World Wide Web; Information retrieval; Static web page; Matching (statistics); Element (criminal law); Web navigation; Programming language","score_opus":0.04308156580098555,"score_gpt":0.32803185587840494,"score_spread":0.2849502900774194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118960523","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03164152,0.00013451131,0.952106,0.00018555496,0.000079158795,0.00024842058,0.00041705937,0.0081494525,0.007038247],"genre_scores_gemma":[0.25245556,0.00019284323,0.73504233,0.0000999929,0.00008483844,0.00047104806,0.0014185513,0.0037139736,0.0065209256],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.997398,0.0008775657,0.00014537484,0.00041233283,0.0010170169,0.0001495984],"domain_scores_gemma":[0.98405206,0.009813795,0.0009617593,0.0024749632,0.0024759255,0.00022149784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020405108,0.001051658,0.00093865016,0.0024506825,0.0009593972,0.0020201162,0.0013064338,0.0009577008,0.008447116],"category_scores_gemma":[0.02220172,0.0007008631,0.0006426735,0.0015557448,0.0009379484,0.0038139792,0.002472589,0.0011916624,0.003344647],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007899335,0.00055588735,0.0057451273,0.00097548607,0.00005348463,0.0007367812,0.0031763057,0.04505501,0.09743034,0.10425052,0.019749139,0.72148204],"study_design_scores_gemma":[0.00019157781,0.00039387227,0.0020988188,0.0002379538,0.00012568437,0.0004937976,0.00077240093,0.52957284,0.20427337,0.15672453,0.104978785,0.00013642883],"about_ca_topic_score_codex":0.0006276291,"about_ca_topic_score_gemma":0.0011723266,"teacher_disagreement_score":0.008447116,"about_ca_system_score_codex":0.00083398574,"about_ca_system_score_gemma":0.0009888714,"threshold_uncertainty_score":0.028258383},"labels":[],"label_agreement":null},{"id":"W2122427912","doi":"10.1023/b:mind.0000045987.92742.71","title":"Inductive Reasoning and Chance Discovery","year":2004,"lang":"en","type":"article","venue":"Minds and Machines","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Surprise; Computer science; Inductive reasoning; Psychology of reasoning; Bayesian probability; Artificial intelligence; Theory of computation; Dilemma; Planner; Machine learning; Model-based reasoning; Epistemology; Psychology; Algorithm; Knowledge representation and reasoning; Philosophy","score_opus":0.005800492898316891,"score_gpt":0.25020309100197474,"score_spread":0.24440259810365786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122427912","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03194647,0.0062630856,0.87641686,0.021601815,0.0004371742,0.00007234697,0.00030758226,0.0002502376,0.06270443],"genre_scores_gemma":[0.8248872,0.004237128,0.14774396,0.0018867586,0.0024427082,0.00035521574,0.0005465047,0.00014261648,0.01775785],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9920272,0.0046868264,0.00033440674,0.0010832159,0.0014981882,0.00037009662],"domain_scores_gemma":[0.93819743,0.05542695,0.0018297702,0.0025127355,0.0014060624,0.00062712136],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009940317,0.00079963007,0.0016469056,0.0038430586,0.0022164192,0.0043589333,0.0027500442,0.0025345243,0.008696195],"category_scores_gemma":[0.045087434,0.0008474846,0.0020553225,0.003351704,0.009169532,0.010986322,0.0043667173,0.004683654,0.0008938144],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014045532,0.000007764732,0.0002968845,0.000037995233,0.00003120164,0.000040533138,0.000120478166,0.0015031485,0.000030257677,0.9910625,0.00091921247,0.00593598],"study_design_scores_gemma":[0.0000027888425,9.709054e-7,0.000036248337,0.000005211367,0.0000039896436,0.0000139907415,0.000009488845,0.0019671086,0.000025882357,0.99734735,0.0005850392,0.000001939602],"about_ca_topic_score_codex":0.0013828921,"about_ca_topic_score_gemma":0.0012122605,"teacher_disagreement_score":0.009940317,"about_ca_system_score_codex":0.0021309117,"about_ca_system_score_gemma":0.0012895302,"threshold_uncertainty_score":0.052570045},"labels":[],"label_agreement":null},{"id":"W2123409753","doi":"10.1111/0824-7935.00126","title":"Probability‐Based Chinese Text Processing and Retrieval","year":2000,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"University of Waterloo; University of Regina","keywords":"Weighting; Computer science; Artificial intelligence; Natural language processing; Probabilistic logic; Word (group theory); Text processing; Term (time); Focus (optics); Relevance (law); Pattern recognition (psychology); Speech recognition; Information retrieval; Mathematics","score_opus":0.027998545217532894,"score_gpt":0.31411299535793963,"score_spread":0.2861144501404067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123409753","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0151883,0.0022439433,0.97127986,0.00042547614,0.00027959637,0.0005387391,0.0004890904,0.0023280552,0.007227033],"genre_scores_gemma":[0.22097382,0.003743188,0.7574415,0.00040933033,0.00050471124,0.0010118982,0.0021219025,0.00031797093,0.013475644],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981535,0.00038216612,0.00022367395,0.00029446618,0.0008258056,0.00012034368],"domain_scores_gemma":[0.9978517,0.00081067876,0.00021062497,0.00031187455,0.0007615825,0.000053493775],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017236137,0.0009585392,0.0009908661,0.0044031823,0.001054592,0.0016279,0.0013216145,0.00058295444,0.008137869],"category_scores_gemma":[0.0071096313,0.0003382202,0.0013088058,0.0054359436,0.00078530586,0.0038422388,0.00085727876,0.00056266435,0.004258061],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021681009,0.00014539376,0.001796992,0.0010826536,0.00018546978,0.00024985883,0.00033168762,0.02328151,0.027449714,0.04037834,0.015223554,0.88965803],"study_design_scores_gemma":[0.00014399111,0.00059379626,0.01014298,0.0001334539,0.0005032263,0.0016298048,0.00027532823,0.7218625,0.12539871,0.049393114,0.08960221,0.0003209371],"about_ca_topic_score_codex":0.006251512,"about_ca_topic_score_gemma":0.0046314104,"teacher_disagreement_score":0.008137869,"about_ca_system_score_codex":0.001355403,"about_ca_system_score_gemma":0.0015639933,"threshold_uncertainty_score":0.027223945},"labels":[],"label_agreement":null},{"id":"W2128027916","doi":"","title":"University of Waterloo at TREC 2008 Blog track","year":2008,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Lexicon; Divergence (linguistics); Information retrieval; Natural language processing; Matching (statistics); Track (disk drive); Artificial intelligence; Polarity (international relations); Linguistics; Statistics; Mathematics","score_opus":0.030359246017418007,"score_gpt":0.23594040493848395,"score_spread":0.20558115892106593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128027916","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06318483,0.008305123,0.012526862,0.031609952,0.0054027913,0.0027140516,0.401353,0.01544599,0.45945746],"genre_scores_gemma":[0.08645991,0.0031636541,0.022779083,0.0024129474,0.000774231,0.0009803488,0.39273846,0.0016682595,0.4890231],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99709153,0.0004257826,0.0001238317,0.000495307,0.0015312867,0.0003321784],"domain_scores_gemma":[0.9902734,0.0008415773,0.00022078025,0.00056073535,0.0069873654,0.0011161343],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003924493,0.00091375184,0.0010965926,0.0035153732,0.0039269566,0.004647271,0.0012181875,0.0008283218,0.06664876],"category_scores_gemma":[0.00715332,0.0005189852,0.00026742645,0.0032678563,0.00071255973,0.0029928903,0.0011178502,0.0011139951,0.026151825],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007921858,0.000101853315,0.001245736,0.00014421693,0.000008911351,0.000035235433,0.00012672295,0.00015820065,0.0016063503,0.0006836491,0.9637947,0.032015137],"study_design_scores_gemma":[0.00017495161,0.00015320693,0.019710366,0.00017511185,0.00003182698,0.000066844455,0.00070185994,0.0051787486,0.0053416416,0.0011242372,0.96724606,0.00009504301],"about_ca_topic_score_codex":0.47273383,"about_ca_topic_score_gemma":0.682788,"teacher_disagreement_score":0.47273383,"about_ca_system_score_codex":0.009687452,"about_ca_system_score_gemma":0.013019221,"threshold_uncertainty_score":0.9399644},"labels":[],"label_agreement":null},{"id":"W2129647599","doi":"10.1145/2232817.2232852","title":"AckSeer","year":2012,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; U.S. Air Force; Deutsche Forschungsgemeinschaft; National Aeronautics and Space Administration; U.S. Department of Energy; Office of the Dean for Research, Princeton University; National Science Foundation","keywords":"Computer science; Search engine indexing; Metadata; Information retrieval; Index (typography); Digital library; Gratitude; Information extraction; World Wide Web; Architecture","score_opus":0.012473141275657124,"score_gpt":0.27952524295426245,"score_spread":0.2670521016786053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129647599","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015187016,0.016775973,0.0892698,0.006704552,0.006980689,0.0014167815,0.380922,0.16700615,0.31573704],"genre_scores_gemma":[0.045821104,0.008033125,0.1176475,0.0027586794,0.0035372723,0.00075640134,0.4655723,0.035267737,0.32060584],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99528116,0.0007018388,0.0007722683,0.0006339793,0.0023712409,0.00023947403],"domain_scores_gemma":[0.9710382,0.010059377,0.004922807,0.005006123,0.007166298,0.0018072402],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.004033694,0.001368514,0.0010450383,0.018155197,0.0013774774,0.0062391567,0.0018768617,0.0015076718,0.17666928],"category_scores_gemma":[0.029929426,0.0005961975,0.00072341354,0.014807951,0.0005837568,0.009207159,0.004095051,0.0010427828,0.13660681],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002971338,0.00003813075,0.0014114474,0.0018475086,0.000040301955,0.00019358391,0.00030137337,0.00019664837,0.0020363976,0.0074600624,0.75891536,0.22726204],"study_design_scores_gemma":[0.000025696705,0.000036031193,0.0011980728,0.00023642399,0.000013436192,0.00021685247,0.00013719339,0.00048370825,0.0016863896,0.0022069204,0.99373347,0.000025759173],"about_ca_topic_score_codex":0.0017540775,"about_ca_topic_score_gemma":0.0030924676,"teacher_disagreement_score":0.8233307,"about_ca_system_score_codex":0.0009172807,"about_ca_system_score_gemma":0.002086539,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2131347559","doi":"10.4018/978-1-60960-625-1.ch004","title":"A Cognitive-Based Approach to Identify Topics in Text Using the Web as a Knowledge Source","year":2011,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Computer science; Meaning (existential); Identification (biology); Ambiguity; Representation (politics); Cluster analysis; Artificial intelligence; Knowledge extraction; Natural language processing; Information retrieval; Psychology","score_opus":0.053387598942272685,"score_gpt":0.32637386018812914,"score_spread":0.27298626124585645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2131347559","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009422889,0.0045088045,0.9254574,0.0029499808,0.00025885875,0.0002744972,0.00041571312,0.0010350281,0.055676892],"genre_scores_gemma":[0.10963186,0.004498694,0.8606618,0.00071703375,0.00046068226,0.00054323376,0.00092091435,0.00018432098,0.02238155],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990165,0.00026342375,0.000056897916,0.00024097806,0.0003724217,0.000049776078],"domain_scores_gemma":[0.9983912,0.0011482849,0.00010139586,0.00012834885,0.00017746058,0.00005323391],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014773386,0.0009882564,0.0004946273,0.007052731,0.0014305661,0.006144649,0.0021084112,0.0012347202,0.0066517764],"category_scores_gemma":[0.0039866925,0.00034996736,0.0013510752,0.0046746708,0.0031756593,0.0066359434,0.0019326222,0.0014869948,0.002434456],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007203802,0.0001612693,0.0012911372,0.0011794118,0.000115192655,0.0003560325,0.007226617,0.004417319,0.008780436,0.37940133,0.015411729,0.58158755],"study_design_scores_gemma":[0.000036101977,0.00006762933,0.0035779106,0.0005361459,0.00014262467,0.0009397676,0.0043534636,0.06395014,0.0072333254,0.6963427,0.22271778,0.0001023982],"about_ca_topic_score_codex":0.0033194702,"about_ca_topic_score_gemma":0.0045667663,"teacher_disagreement_score":0.007052731,"about_ca_system_score_codex":0.0017231953,"about_ca_system_score_gemma":0.0016602892,"threshold_uncertainty_score":0.02225244},"labels":[],"label_agreement":null},{"id":"W2133556763","doi":"10.1007/3-540-47922-8_14","title":"Topic Discovery from Text Using Aggregation of Different Clustering Methods","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Cluster analysis; Computer science; Document clustering; Measure (data warehouse); Clustering high-dimensional data; Data mining; Process (computing); Cluster (spacecraft); Vocabulary; Artificial intelligence; Correlation clustering; Machine learning; Information retrieval","score_opus":0.03332615904962769,"score_gpt":0.3142565827528983,"score_spread":0.2809304237032706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2133556763","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0917286,0.003619302,0.8905331,0.0004455349,0.0004352691,0.0006850226,0.0031533288,0.0063946014,0.0030051805],"genre_scores_gemma":[0.20864116,0.0011113514,0.77503055,0.000098363234,0.00050263753,0.00050821673,0.009695867,0.00069290685,0.003719025],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954222,0.0010662035,0.0005107737,0.0012759604,0.0013420472,0.0003828142],"domain_scores_gemma":[0.99063045,0.0046436614,0.00042982373,0.0013199412,0.002638512,0.00033754532],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048128897,0.0015615709,0.0022679542,0.014453964,0.0018063217,0.003806309,0.0020000392,0.0015993052,0.0019191918],"category_scores_gemma":[0.012228522,0.0006508968,0.00283672,0.013726076,0.00048437237,0.0029748403,0.0024602965,0.0012402772,0.0020478142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013706306,0.000431924,0.0135754505,0.00091039756,0.00091743586,0.00021677242,0.0014854845,0.010210137,0.03961224,0.0022266633,0.012404026,0.9166389],"study_design_scores_gemma":[0.00033457766,0.0006283208,0.044186212,0.00024224512,0.0026654552,0.0010267409,0.0022427666,0.809357,0.07135933,0.037249316,0.030327778,0.0003802708],"about_ca_topic_score_codex":0.0044249524,"about_ca_topic_score_gemma":0.009474416,"teacher_disagreement_score":0.014453964,"about_ca_system_score_codex":0.00077877694,"about_ca_system_score_gemma":0.0012457347,"threshold_uncertainty_score":0.02545333},"labels":[],"label_agreement":null},{"id":"W2134878975","doi":"10.1016/j.ijinfomgt.2010.10.003","title":"Making functional units functional: The role of rhetorical structure in use of scholarly journal articles","year":2010,"lang":"en","type":"article","venue":"International Journal of Information Management","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reading (process); Computer science; Relevance (law); Set (abstract data type); Taxonomy (biology); Rhetorical question; Function (biology); Information retrieval; Linguistics; Political science","score_opus":0.032068415172043674,"score_gpt":0.2803225324927541,"score_spread":0.2482541173207104,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134878975","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8261103,0.0011510194,0.064973965,0.006960547,0.00024807695,0.00017961975,0.000245564,0.00047917923,0.099651754],"genre_scores_gemma":[0.9893819,0.00014619784,0.008991879,0.00015888222,0.00011486062,0.00005710522,0.00009207525,0.00016760339,0.00088943826],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9808919,0.012459768,0.001119941,0.0013221671,0.0033120902,0.000894078],"domain_scores_gemma":[0.8105189,0.1479999,0.015976159,0.011555511,0.010254246,0.0036953324],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.018487219,0.0005583224,0.0005020124,0.0046463953,0.00302755,0.010451668,0.0023332015,0.0027426863,0.004683996],"category_scores_gemma":[0.15838088,0.0008485674,0.0004649103,0.002860704,0.0072842455,0.017255178,0.004502052,0.002366124,0.001085374],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016171715,0.0007504344,0.07796485,0.0013950226,0.00026505985,0.0009467911,0.14886092,0.0030791305,0.03822656,0.3862243,0.005177625,0.3354922],"study_design_scores_gemma":[0.00039803886,0.0013011063,0.16696063,0.0013427023,0.0006931243,0.0011312392,0.06797324,0.031900305,0.0299456,0.6531636,0.044904727,0.00028565215],"about_ca_topic_score_codex":0.0013109653,"about_ca_topic_score_gemma":0.0015135817,"teacher_disagreement_score":0.9895483,"about_ca_system_score_codex":0.001736159,"about_ca_system_score_gemma":0.0037586137,"threshold_uncertainty_score":0.09777093},"labels":[],"label_agreement":null},{"id":"W2137779158","doi":"10.3115/1654758.1654765","title":"A study of two graph algorithms in topic-driven summarization","year":2006,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Automatic summarization; Computer science; Sentence; Graph; Natural language processing; Information retrieval; Artificial intelligence; Scope (computer science); Matching (statistics); Algorithm; Theoretical computer science; Mathematics","score_opus":0.012980108970599475,"score_gpt":0.29398154912421026,"score_spread":0.2810014401536108,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2137779158","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044798605,0.004365823,0.94222254,0.0015939682,0.00018890014,0.00025971467,0.00011780711,0.0015174001,0.004935256],"genre_scores_gemma":[0.2770757,0.0020089268,0.7158581,0.0005264116,0.00041967272,0.00027795858,0.000634598,0.00072116946,0.0024774605],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99229103,0.004732754,0.00040367353,0.0011392419,0.0011635297,0.00026977016],"domain_scores_gemma":[0.91554713,0.073154755,0.0022271334,0.003590507,0.004798887,0.0006816318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011603529,0.0015715236,0.0016793701,0.004718859,0.0011420187,0.0034347007,0.002336087,0.0034501315,0.0019596228],"category_scores_gemma":[0.06429534,0.00077928026,0.0013625459,0.004997315,0.0017517095,0.007730771,0.0013575915,0.0023372131,0.0006902757],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009646501,0.00077102927,0.007005,0.0012050758,0.00069578853,0.0002533107,0.0017285472,0.22494058,0.012065408,0.12169665,0.008898193,0.61977565],"study_design_scores_gemma":[0.00018684474,0.0006517842,0.002406413,0.00008765007,0.00018843325,0.00025598536,0.0003441463,0.9166101,0.0065800715,0.06613196,0.0064833667,0.0000731691],"about_ca_topic_score_codex":0.004986711,"about_ca_topic_score_gemma":0.0042997287,"teacher_disagreement_score":0.011603529,"about_ca_system_score_codex":0.0018108918,"about_ca_system_score_gemma":0.0012041711,"threshold_uncertainty_score":0.06136608},"labels":[],"label_agreement":null},{"id":"W2138954094","doi":"10.1145/1097047.1097059","title":"Narrative text classification for automatic key phrase extraction in web document corpora","year":2005,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Automatic summarization; Information retrieval; Plain text; Key (lock); Natural language processing; Phrase; tf–idf; Artificial intelligence; Ranking (information retrieval); Document clustering; Information extraction; HTML element; Web page; Keyword extraction; Cluster analysis; World Wide Web; Term (time)","score_opus":0.024613583411413936,"score_gpt":0.3336148891383208,"score_spread":0.30900130572690687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2138954094","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.104035564,0.0029146248,0.8533756,0.0006291355,0.00032798378,0.0019500445,0.010337648,0.02095034,0.0054790503],"genre_scores_gemma":[0.09328279,0.0006168459,0.88523245,0.00007071499,0.00012409707,0.00138948,0.01720149,0.0005766425,0.0015054517],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961916,0.0014049871,0.00083532836,0.0005765585,0.0008472668,0.00014413612],"domain_scores_gemma":[0.9856133,0.008096746,0.0010788334,0.0012645902,0.0037039393,0.0002426057],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049716057,0.0012798693,0.0010247612,0.0070925825,0.0013096656,0.001990597,0.0010435375,0.0009905315,0.006275678],"category_scores_gemma":[0.027712876,0.0004659655,0.0009978147,0.005589753,0.00050086336,0.004242604,0.0009427947,0.0009187756,0.006822853],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007460054,0.00026276562,0.0053785075,0.002387255,0.00015219736,0.00035699914,0.0010786415,0.004414628,0.120073825,0.0037291946,0.023581518,0.8378384],"study_design_scores_gemma":[0.00062496564,0.0016035894,0.045940034,0.00066336483,0.00056811905,0.0026702874,0.0023663864,0.5175076,0.28189915,0.011366278,0.13440453,0.0003857906],"about_ca_topic_score_codex":0.0026030242,"about_ca_topic_score_gemma":0.0035200147,"teacher_disagreement_score":0.0070925825,"about_ca_system_score_codex":0.0007999136,"about_ca_system_score_gemma":0.0014032733,"threshold_uncertainty_score":0.026292682},"labels":[],"label_agreement":null},{"id":"W2139462516","doi":"10.7202/602712ar","title":"Le syntagme nominal : exemple d’un phénomène d’anticipation en lecture","year":2009,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Humanities; Anticipation (artificial intelligence); Art; Philosophy; Computer science; Artificial intelligence","score_opus":0.010455044568281249,"score_gpt":0.28509243079679375,"score_spread":0.2746373862285125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2139462516","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3750544,0.0019687766,0.24148075,0.010130108,0.0015987918,0.00015743102,0.00063683576,0.0020291475,0.36694384],"genre_scores_gemma":[0.9364813,0.0004699074,0.016905352,0.00047461927,0.00012418692,0.00006565865,0.00024082877,0.00037926473,0.044858936],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9982632,0.00087935117,0.000043557262,0.00022746192,0.00047050277,0.000115940646],"domain_scores_gemma":[0.9977398,0.0016122662,0.00010685417,0.00018839622,0.0002690595,0.000083687344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013081564,0.0004660655,0.00023679482,0.0006558591,0.0024717941,0.0026507503,0.0005033121,0.0011580703,0.009218869],"category_scores_gemma":[0.006522521,0.00021866447,0.00025750705,0.00089338433,0.0038429382,0.004290037,0.0019695358,0.0024547928,0.0014668991],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005260206,0.0000408043,0.0017618492,0.00031302174,0.000014785345,0.0018221182,0.21188122,0.00083668204,0.020038282,0.6982223,0.01197829,0.052564677],"study_design_scores_gemma":[0.00008754004,0.00024440294,0.0062941154,0.0003154552,0.000037883026,0.0044946875,0.113473855,0.007940801,0.017394818,0.10612614,0.7434141,0.00017622428],"about_ca_topic_score_codex":0.006065494,"about_ca_topic_score_gemma":0.0064184554,"teacher_disagreement_score":0.009218869,"about_ca_system_score_codex":0.0019861083,"about_ca_system_score_gemma":0.0009217686,"threshold_uncertainty_score":0.030840218},"labels":[],"label_agreement":null},{"id":"W2140061837","doi":"10.16995/dscn.253","title":"Toward Next Generation Text Analysis Tools: The Text Analysis Markup Language (TAML)","year":2005,"lang":"en","type":"article","venue":"Digital Studies / Le champ numérique","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Markup language; Computer science; Interoperability; Standardization; World Wide Web; Vocabulary; Linguistics; XML","score_opus":0.05723028432510359,"score_gpt":0.29959740054323974,"score_spread":0.24236711621813617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140061837","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005113865,0.0010811277,0.9508052,0.01053968,0.00067801797,0.0002517549,0.0022302214,0.019892454,0.009407742],"genre_scores_gemma":[0.035049766,0.0013394381,0.92947876,0.002423587,0.0005694105,0.00038857743,0.0059443954,0.0055430955,0.019262936],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9930321,0.003162756,0.0009871358,0.0008963492,0.0016610945,0.00026049698],"domain_scores_gemma":[0.92687774,0.034167353,0.004397001,0.01198754,0.019732228,0.0028380866],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018066438,0.0008788866,0.00090999925,0.0032098028,0.0011544824,0.008604689,0.0030698779,0.0017494763,0.018024351],"category_scores_gemma":[0.055379696,0.00073459494,0.0010019868,0.0021117476,0.0017507826,0.015298329,0.0037274323,0.0029616486,0.021158487],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038027094,0.00019178029,0.0031480698,0.0016928741,0.00006964733,0.00031481838,0.0021127453,0.00084302854,0.034863368,0.1249069,0.0967388,0.73473775],"study_design_scores_gemma":[0.00008944813,0.00026055018,0.001626972,0.001776604,0.00013299343,0.0008161377,0.0014482965,0.03516026,0.0808196,0.10670091,0.7709618,0.00020636882],"about_ca_topic_score_codex":0.0009306352,"about_ca_topic_score_gemma":0.00087545806,"teacher_disagreement_score":0.018066438,"about_ca_system_score_codex":0.0012330665,"about_ca_system_score_gemma":0.00357541,"threshold_uncertainty_score":0.09554559},"labels":[],"label_agreement":null},{"id":"W2140162241","doi":"10.1007/s11192-011-0589-1","title":"Author disambiguation using multi-aspect similarity indicators","year":2011,"lang":"en","type":"article","venue":"Scientometrics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Genome Canada","keywords":"Similarity (geometry); Recall; Precision and recall; Computer science; Task (project management); Measure (data warehouse); Information retrieval; Key (lock); Data mining; Artificial intelligence; Natural language processing; Psychology; Cognitive psychology","score_opus":0.20036793714287734,"score_gpt":0.3850923397812263,"score_spread":0.18472440263834897,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140162241","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2593812,0.00233163,0.71416986,0.0005065639,0.0002674919,0.00052259414,0.0043134317,0.009008055,0.009499171],"genre_scores_gemma":[0.5136863,0.0007228502,0.47793403,0.000044933266,0.00018256936,0.00028010109,0.0050258627,0.00042561052,0.0016978487],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99223095,0.0015767596,0.0014200117,0.0014213687,0.0030043693,0.00034656475],"domain_scores_gemma":[0.9783392,0.009682412,0.0040027197,0.0027177758,0.004802476,0.00045540812],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.005651432,0.0011261646,0.0017810812,0.038126208,0.0013586935,0.0049274648,0.00117234,0.0012350989,0.0016674418],"category_scores_gemma":[0.037942942,0.00042952385,0.0012629132,0.032639205,0.00058969023,0.0054242057,0.0029008149,0.0008909502,0.0019427014],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000495093,0.00028547886,0.0906014,0.00076293555,0.00040016577,0.00032704195,0.0011640802,0.014824714,0.014622401,0.009140316,0.006980726,0.8603957],"study_design_scores_gemma":[0.00020040046,0.000609591,0.14619578,0.00033955916,0.00054563873,0.0024608728,0.0023563018,0.6328462,0.09661568,0.067694925,0.049699243,0.000435803],"about_ca_topic_score_codex":0.001413587,"about_ca_topic_score_gemma":0.0024078325,"teacher_disagreement_score":0.96187377,"about_ca_system_score_codex":0.0008656283,"about_ca_system_score_gemma":0.0015976714,"threshold_uncertainty_score":0.029887974},"labels":[],"label_agreement":null},{"id":"W2141430397","doi":"10.1504/ijcat.2010.034534","title":"A schema for ontology-based concept definition and identification","year":2010,"lang":"en","type":"article","venue":"International Journal of Computer Applications in Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Schema (genetic algorithms); Information retrieval; Semantic Web; Ontology; World Wide Web","score_opus":0.01012021903614982,"score_gpt":0.31089422084601154,"score_spread":0.3007740018098617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141430397","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015259403,0.00044572505,0.981736,0.00083021395,0.00027001536,0.0007970536,0.0032070056,0.0029296,0.008258486],"genre_scores_gemma":[0.010608548,0.00052790163,0.97768766,0.0003587532,0.000073818584,0.0009797732,0.006229189,0.0003390743,0.0031952905],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9950322,0.0011494443,0.0013596533,0.00087590836,0.0013990863,0.00018370511],"domain_scores_gemma":[0.9951728,0.0010444544,0.00057664845,0.0012644294,0.0016654684,0.0002761367],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006008754,0.0013305816,0.0011702579,0.0044709137,0.0019906813,0.0053836997,0.0034371156,0.002407732,0.007545362],"category_scores_gemma":[0.009859046,0.0011362615,0.0027320776,0.005741698,0.0015987523,0.00898458,0.0031078912,0.0046424232,0.0049624215],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012492934,0.00032593808,0.002055262,0.0008939242,0.00019772546,0.0006050757,0.0027520824,0.0057812296,0.011864695,0.67156696,0.055677515,0.24815468],"study_design_scores_gemma":[0.000059014317,0.00009280449,0.0013171831,0.00049395714,0.00012644577,0.0018795071,0.00082972506,0.039010614,0.006995424,0.18669066,0.7623749,0.00012985608],"about_ca_topic_score_codex":0.0067458972,"about_ca_topic_score_gemma":0.006038225,"teacher_disagreement_score":0.007545362,"about_ca_system_score_codex":0.0017647847,"about_ca_system_score_gemma":0.0056748996,"threshold_uncertainty_score":0.03177774},"labels":[],"label_agreement":null},{"id":"W2142243548","doi":"10.48550/arxiv.1204.2847","title":"Segmentation Similarity and Agreement","year":2012,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Segmentation; Similarity (geometry); Metric (unit); Computer science; Scale-space segmentation; Artificial intelligence; Edit distance; Pattern recognition (psychology); Image segmentation; Agreement; Segmentation-based object categorization; Image (mathematics)","score_opus":0.058679598677329574,"score_gpt":0.2023831825991733,"score_spread":0.1437035839218437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142243548","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.100942366,0.0018737068,0.87081856,0.00040576898,0.00042599434,0.00052395003,0.0013006376,0.00202291,0.02168619],"genre_scores_gemma":[0.7063667,0.00050693296,0.2835647,0.00027866557,0.00022915893,0.0006684719,0.0028505437,0.0014038724,0.004130984],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9339591,0.021727176,0.007696694,0.01291866,0.021918638,0.0017797174],"domain_scores_gemma":[0.8541687,0.07668695,0.01268672,0.01784084,0.036439016,0.0021777318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.029221356,0.0014088658,0.0019390003,0.010390236,0.0018172364,0.005890527,0.0029303576,0.0030212554,0.004936952],"category_scores_gemma":[0.14777778,0.000831846,0.0018440833,0.0069625694,0.0042290464,0.009738747,0.006161086,0.0022083563,0.0028214506],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029732708,0.00032936558,0.10065724,0.0027808184,0.0018297661,0.000732849,0.00995884,0.055148188,0.069305606,0.06519295,0.018718272,0.6723728],"study_design_scores_gemma":[0.00022077662,0.0017832507,0.13134038,0.0007319716,0.0011367762,0.003607615,0.0063129663,0.44307992,0.11876989,0.2323204,0.059741586,0.00095447886],"about_ca_topic_score_codex":0.0019009969,"about_ca_topic_score_gemma":0.0024852762,"teacher_disagreement_score":0.029221356,"about_ca_system_score_codex":0.0020021172,"about_ca_system_score_gemma":0.0015395867,"threshold_uncertainty_score":0.15453917},"labels":[],"label_agreement":null},{"id":"W2147695060","doi":"10.13053/cys-17-2-1523","title":"A Knowledge-Base Oriented Approach for Automatic Keyword Extraction","year":2013,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal; Polytechnique Montréal","funders":"","keywords":"Keyword extraction; Computer science; Information retrieval; Rank (graph theory); Novelty; Task (project management); Process (computing); Keyword density; Knowledge base; Artificial intelligence; Data mining; Natural language processing; Keyword search; Mathematics","score_opus":0.02112686952801224,"score_gpt":0.30501400958426567,"score_spread":0.28388714005625343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2147695060","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0063699097,0.0011496281,0.954098,0.0004577905,0.0001393935,0.00057138223,0.0032661136,0.027791282,0.006156492],"genre_scores_gemma":[0.040160935,0.00076792494,0.9467275,0.00036295058,0.000116439456,0.00034041773,0.0069473437,0.00051661837,0.004059825],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99808925,0.00028506,0.00031449323,0.0004988095,0.00068688486,0.00012548843],"domain_scores_gemma":[0.99563473,0.0014288252,0.00039188433,0.00067539344,0.0016679108,0.00020120447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019286649,0.0013539877,0.0016203908,0.012190434,0.0013596676,0.0033511193,0.0025258511,0.0016890214,0.007513069],"category_scores_gemma":[0.008649426,0.0006285045,0.0011576016,0.007538524,0.00063897856,0.0043740473,0.0021939832,0.0015731421,0.012567558],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002475813,0.0003157574,0.0015080998,0.0010164084,0.00019852968,0.0005675465,0.00033770426,0.0028639415,0.056107983,0.0059020906,0.03361746,0.89731675],"study_design_scores_gemma":[0.00023790382,0.0006132715,0.0075458335,0.00076812407,0.0009714356,0.0052142097,0.0014410365,0.42158368,0.22650711,0.054041553,0.28060034,0.00047558648],"about_ca_topic_score_codex":0.005033051,"about_ca_topic_score_gemma":0.0077712657,"teacher_disagreement_score":0.012190434,"about_ca_system_score_codex":0.0010441671,"about_ca_system_score_gemma":0.0021114182,"threshold_uncertainty_score":0.025133729},"labels":[],"label_agreement":null},{"id":"W2148404145","doi":"10.3115/v1/d14-1168","title":"Abstractive Summarization of Product Reviews Using Discourse Structure","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":200,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Automatic summarization; Computer science; Natural language processing; Product (mathematics); Linguistics; Artificial intelligence; Mathematics; Philosophy","score_opus":0.019390900814182644,"score_gpt":0.326621961672267,"score_spread":0.3072310608580843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2148404145","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033097,0.0017089252,0.93829703,0.0009850557,0.00028312337,0.00069863256,0.003331871,0.017625393,0.003973037],"genre_scores_gemma":[0.11329141,0.0011427234,0.8689172,0.0002075729,0.00039963398,0.00046737902,0.010572516,0.0007954981,0.004205952],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977144,0.00072148565,0.00025337664,0.00055536214,0.0006858599,0.000069544614],"domain_scores_gemma":[0.9927004,0.0033347055,0.0011233699,0.00057709886,0.0021371657,0.0001272358],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001897499,0.0018840276,0.00130155,0.006441223,0.00078751246,0.0029538204,0.0013234181,0.0009060662,0.0029571252],"category_scores_gemma":[0.010801622,0.00064053363,0.001068269,0.0030653572,0.00039349738,0.0038512254,0.0014042532,0.0011860862,0.002579065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030252247,0.0001741771,0.0015465358,0.0018276161,0.00021687568,0.0004304475,0.0025666612,0.013407357,0.05523998,0.006826086,0.025384007,0.8920778],"study_design_scores_gemma":[0.00021706552,0.0006730478,0.0070320354,0.0005131169,0.00097617094,0.00078768795,0.0025762585,0.6140401,0.14265195,0.042545468,0.18770678,0.0002803969],"about_ca_topic_score_codex":0.002387718,"about_ca_topic_score_gemma":0.0033255212,"teacher_disagreement_score":0.006441223,"about_ca_system_score_codex":0.0007580305,"about_ca_system_score_gemma":0.0012932215,"threshold_uncertainty_score":0.010035038},"labels":[],"label_agreement":null},{"id":"W2148945625","doi":"10.19173/irrodl.v14i1.1389","title":"Automatic evaluation for e-learning using latent semantic analysis: A use case","year":2013,"lang":"en","type":"article","venue":"The International Review of Research in Open and Distributed Learning","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Universitat Oberta de Catalunya; European Commission","keywords":"Animal science; Biology","score_opus":0.184779399202003,"score_gpt":0.4953805839448462,"score_spread":0.31060118474284315,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2148945625","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3213757,0.0038642923,0.5062026,0.0030147291,0.00093674,0.001951257,0.017291363,0.11552644,0.029836962],"genre_scores_gemma":[0.62107223,0.00070480764,0.33189556,0.00042643404,0.00008967668,0.0006706121,0.028882887,0.0017586625,0.014499134],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9916863,0.003332732,0.00075362675,0.0014924605,0.0022867653,0.0004480795],"domain_scores_gemma":[0.98838824,0.0065186475,0.0002858182,0.0021863761,0.0021453844,0.0004755061],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007934169,0.0016202929,0.0011833243,0.0022966953,0.00096114055,0.0033996934,0.002390364,0.0031391988,0.013772288],"category_scores_gemma":[0.01959378,0.00049217464,0.0012251723,0.0018073105,0.00077481783,0.0055606067,0.0026622494,0.0018082681,0.008817179],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003596976,0.003593033,0.0105825495,0.0014344451,0.00042814552,0.0014836324,0.0004988311,0.04909708,0.014261905,0.0077373027,0.066988245,0.8402978],"study_design_scores_gemma":[0.00042870213,0.00063272147,0.0042737587,0.00017538649,0.00012592525,0.0005248895,0.00057497446,0.92554075,0.034104228,0.0075915167,0.02594678,0.00008048133],"about_ca_topic_score_codex":0.009634626,"about_ca_topic_score_gemma":0.008880126,"teacher_disagreement_score":0.013772288,"about_ca_system_score_codex":0.0017130125,"about_ca_system_score_gemma":0.0017369865,"threshold_uncertainty_score":0.04607284},"labels":[],"label_agreement":null},{"id":"W2149120305","doi":"10.21437/interspeech.2007-66","title":"The voice-rate dialog system for consumer ratings","year":2007,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Dialog box; Phone; Computer science; Dialog system; Product (mathematics); Matching (statistics); Speech recognition; Key (lock); Quarter (Canadian coin); World Wide Web; Computer security","score_opus":0.0127989999599461,"score_gpt":0.28152221592549637,"score_spread":0.26872321596555027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149120305","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0056209504,0.0010720796,0.3885725,0.0006456373,0.0007832129,0.003149823,0.0733897,0.4630126,0.06375347],"genre_scores_gemma":[0.09715621,0.0009681027,0.5576102,0.0012153834,0.0010360109,0.0077530616,0.18132155,0.032702778,0.12023675],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955967,0.001604597,0.00045149718,0.0005381944,0.0016132541,0.00019575248],"domain_scores_gemma":[0.9941724,0.0018204933,0.000295735,0.0013374891,0.001944745,0.00042922297],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004854901,0.0015998465,0.0016051495,0.0019152645,0.00084538833,0.0024954244,0.002266152,0.0016724151,0.12546071],"category_scores_gemma":[0.016314289,0.0009266744,0.0006551538,0.0017653679,0.00038841364,0.0036846166,0.0030755028,0.0014458225,0.10934029],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015030843,0.0001932294,0.0011553423,0.0008428113,0.00007605156,0.00024045653,0.00040934604,0.0008711737,0.010884579,0.009810285,0.7665712,0.20744246],"study_design_scores_gemma":[0.0010707418,0.0004137588,0.006923337,0.00022846987,0.000084046966,0.0009295362,0.00019099207,0.057318203,0.021912795,0.022899404,0.887598,0.00043067292],"about_ca_topic_score_codex":0.0018943686,"about_ca_topic_score_gemma":0.0029178509,"teacher_disagreement_score":0.12546071,"about_ca_system_score_codex":0.00061884226,"about_ca_system_score_gemma":0.0010872326,"threshold_uncertainty_score":0.4197079},"labels":[],"label_agreement":null},{"id":"W2149639484","doi":"","title":"Clustering Voices in The Waste Land","year":2013,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Cluster analysis; Task (project management); Computer science; Cluster (spacecraft); Segmentation; Artificial intelligence; Natural language processing; Engineering","score_opus":0.01142973882149155,"score_gpt":0.2595858226516054,"score_spread":0.24815608383011384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149639484","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7528372,0.0025115074,0.22023904,0.0014457048,0.00051624526,0.00036150758,0.0029859669,0.003332825,0.015770005],"genre_scores_gemma":[0.84462416,0.0004895206,0.13672131,0.00026633334,0.00021975112,0.00013695199,0.008204474,0.00059973344,0.0087376125],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99892324,0.00029879666,0.00006380231,0.00038310184,0.00021325178,0.00011773693],"domain_scores_gemma":[0.9983191,0.0007710825,0.00013897,0.00021986707,0.00045360997,0.00009726627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007748795,0.00078086887,0.0005466752,0.002896694,0.001255753,0.002090945,0.00074892456,0.0011793497,0.0030570508],"category_scores_gemma":[0.0039165346,0.00025237122,0.00062350655,0.0022275194,0.0010783318,0.0015665254,0.0012348332,0.00089130295,0.0028000867],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020294145,0.00029831316,0.021516373,0.00096066674,0.00022518686,0.0020707306,0.012194659,0.029025104,0.11846629,0.010554583,0.028661331,0.77399737],"study_design_scores_gemma":[0.00024212949,0.00055579183,0.07006697,0.00046713004,0.00031757736,0.0027175844,0.030415146,0.5883058,0.13851024,0.043499503,0.12465081,0.0002513799],"about_ca_topic_score_codex":0.0040555666,"about_ca_topic_score_gemma":0.0078040315,"teacher_disagreement_score":0.0040555666,"about_ca_system_score_codex":0.0007318046,"about_ca_system_score_gemma":0.00048318633,"threshold_uncertainty_score":0.010226905},"labels":[],"label_agreement":null},{"id":"W2150630862","doi":"10.1075/ml.5.1.06baa","title":"A real experiment is a factorial experiment?","year":2010,"lang":"en","type":"article","venue":"The Mental Lexicon","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":89,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Factorial experiment; Mathematics; Statistics; Computer science; Artificial intelligence","score_opus":0.016361447109000178,"score_gpt":0.31600834063477085,"score_spread":0.2996468935257707,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150630862","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.080403335,0.015751528,0.7128447,0.057390112,0.02612697,0.016641153,0.0031174677,0.0032350079,0.08448962],"genre_scores_gemma":[0.38345274,0.0047117374,0.51283115,0.029353384,0.0070388163,0.05521366,0.0010186421,0.00086273625,0.005517154],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.75630856,0.18633246,0.010636354,0.024612797,0.020360392,0.001749403],"domain_scores_gemma":[0.3731064,0.5203212,0.026238188,0.065706104,0.0120758815,0.0025522523],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.22381692,0.0021402377,0.004014705,0.0013882575,0.0026064916,0.0079155,0.0032726198,0.007083857,0.01908097],"category_scores_gemma":[0.456216,0.0017782638,0.0027853928,0.0025190928,0.017498758,0.018753255,0.004050239,0.0055257836,0.0029802378],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.017741054,0.002958094,0.013309217,0.013856794,0.004014832,0.00034247417,0.010741399,0.0033402706,0.008836301,0.65029407,0.04975337,0.22481208],"study_design_scores_gemma":[0.00918514,0.016794086,0.018170537,0.0041899057,0.0023585723,0.0005973397,0.002781973,0.010308455,0.0067178197,0.6411614,0.28674337,0.0009913284],"about_ca_topic_score_codex":0.0008887682,"about_ca_topic_score_gemma":0.00074853963,"teacher_disagreement_score":0.22381692,"about_ca_system_score_codex":0.003968998,"about_ca_system_score_gemma":0.0032813125,"threshold_uncertainty_score":0.9571719},"labels":[],"label_agreement":null},{"id":"W2151674174","doi":"","title":"Multilabel Subject-Based Classification of Poetry","year":2015,"lang":"en","type":"article","venue":"Digital Access to Libraries (Université catholique de Louvain (UCL), l'Université de Namur (UNamur) and the Université Saint-Louis (USL-B))","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Poetry; Computer science; Latent Dirichlet allocation; Artificial intelligence; Context (archaeology); Support vector machine; Set (abstract data type); Style (visual arts); Subject (documents); Natural language processing; Task (project management); Machine learning; Simple (philosophy); Pattern recognition (psychology); Topic model; Linguistics; World Wide Web; Literature; Engineering; Art","score_opus":0.017192345718264498,"score_gpt":0.23033702888413238,"score_spread":0.21314468316586788,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2151674174","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40840748,0.0018226877,0.5539742,0.002067645,0.00072783016,0.00044184466,0.0034475853,0.0034763205,0.02563453],"genre_scores_gemma":[0.8775907,0.00031299618,0.10862437,0.00018881621,0.00046119082,0.00026297616,0.004402311,0.0001743517,0.007982319],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980281,0.0007355881,0.00014503671,0.0004757612,0.00042735183,0.00018813643],"domain_scores_gemma":[0.99421257,0.00311167,0.00050360884,0.0006824092,0.0012345045,0.000255251],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024809677,0.00049562874,0.0006808774,0.006182416,0.0012499486,0.002041754,0.0009112708,0.0010115794,0.0038861209],"category_scores_gemma":[0.008980292,0.00017801805,0.0007799556,0.0025821135,0.0011498026,0.0038365438,0.0015370321,0.0013890022,0.0018352659],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012957348,0.0005999942,0.05333651,0.00059015106,0.00014263063,0.0005612143,0.003691105,0.013184914,0.026765656,0.035022747,0.019506428,0.84530294],"study_design_scores_gemma":[0.000093180235,0.0004976387,0.05751976,0.00022145936,0.00018852332,0.001020169,0.003950586,0.6807097,0.026489327,0.18928114,0.039869837,0.00015868973],"about_ca_topic_score_codex":0.0015485589,"about_ca_topic_score_gemma":0.0026985467,"teacher_disagreement_score":0.006182416,"about_ca_system_score_codex":0.0011124776,"about_ca_system_score_gemma":0.00079068384,"threshold_uncertainty_score":0.01312077},"labels":[],"label_agreement":null},{"id":"W2152267628","doi":"10.1109/grc.2006.1635894","title":"The STP model for solving imprecise problems","year":2006,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Computer science; Matching (statistics); Process (computing); Optimization problem; Problem statement; Mathematical optimization; Mathematics; Algorithm; Management science","score_opus":0.014803621059243415,"score_gpt":0.2621537197575264,"score_spread":0.24735009869828298,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2152267628","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008002661,0.00019240356,0.985283,0.0009700847,0.0000441915,0.00014649316,0.00015339018,0.00027538495,0.0049323332],"genre_scores_gemma":[0.17731872,0.00044222942,0.8166471,0.00021736458,0.000112119895,0.000590228,0.00055493397,0.00013467262,0.003982567],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9918052,0.0029572535,0.00087782653,0.0011748667,0.002771006,0.00041379908],"domain_scores_gemma":[0.98636657,0.00902337,0.0011769518,0.001454215,0.0015461505,0.0004327617],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051356996,0.0011685375,0.0014687876,0.002393075,0.0012361094,0.0049841683,0.0032521975,0.0027055538,0.0066374615],"category_scores_gemma":[0.022029804,0.0006638944,0.0026238745,0.003513154,0.0038371524,0.009357464,0.0039455397,0.0049304673,0.000907289],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007388596,0.00010421548,0.00047292493,0.0002885624,0.00006620922,0.00011522265,0.00042400742,0.16572526,0.00078474334,0.7862436,0.0024564627,0.04324491],"study_design_scores_gemma":[0.000028594317,0.000045716082,0.00006255492,0.000056243378,0.000021794463,0.000071495444,0.000117704185,0.46798807,0.00084144145,0.5257641,0.0049859844,0.000016355974],"about_ca_topic_score_codex":0.0043416023,"about_ca_topic_score_gemma":0.003445166,"teacher_disagreement_score":0.0066374615,"about_ca_system_score_codex":0.002611071,"about_ca_system_score_gemma":0.0030481,"threshold_uncertainty_score":0.027160525},"labels":[],"label_agreement":null},{"id":"W2153207410","doi":"10.1145/1978942.1979167","title":"Review spotlight","year":2011,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":83,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Product (mathematics); Adjective; Quality (philosophy); World Wide Web; Noun; Information retrieval; Natural language processing","score_opus":0.040886315015554765,"score_gpt":0.28119789087673647,"score_spread":0.2403115758611817,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2153207410","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033578422,0.012936796,0.35167155,0.0038519895,0.00623992,0.008969997,0.03848661,0.3794043,0.16486047],"genre_scores_gemma":[0.16620716,0.0061536115,0.40564024,0.0042420677,0.0055332403,0.008022251,0.048690427,0.042018548,0.3134924],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99792224,0.0005485934,0.00023516623,0.00043667262,0.00073709246,0.00012023814],"domain_scores_gemma":[0.9713597,0.009660826,0.0020304085,0.004009534,0.010622501,0.0023171392],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0035070614,0.0012811042,0.0014427176,0.004890063,0.000906232,0.0025505521,0.0017863767,0.0009771112,0.11887533],"category_scores_gemma":[0.020366613,0.00080073724,0.00075245596,0.0019465978,0.00030140267,0.0029128003,0.0020082672,0.00091966847,0.07234846],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013463235,0.000091742615,0.0015964266,0.0020758505,0.00009371791,0.00038307585,0.0005440819,0.00018014827,0.014686626,0.0018051211,0.60608006,0.3711168],"study_design_scores_gemma":[0.0002905457,0.0004456285,0.0048352573,0.00029543057,0.00011973962,0.0010887991,0.00025533224,0.0049264715,0.022869507,0.002599622,0.9621205,0.00015314143],"about_ca_topic_score_codex":0.0006226579,"about_ca_topic_score_gemma":0.0013974992,"teacher_disagreement_score":0.8811247,"about_ca_system_score_codex":0.0004718191,"about_ca_system_score_gemma":0.0011590547,"threshold_uncertainty_score":0.3976776},"labels":[],"label_agreement":null},{"id":"W2161427841","doi":"","title":"Summarizing Emails with Conversational Cohesion and Subjectivity","year":2008,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Automatic summarization; Cohesion (chemistry); Computer science; PageRank; Cosine similarity; Natural language processing; Artificial intelligence; Graph; Sentence; Empirical research; Information retrieval; Subjectivity; Theoretical computer science; Pattern recognition (psychology); Mathematics","score_opus":0.013367564952236293,"score_gpt":0.24368251737275914,"score_spread":0.23031495242052286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2161427841","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11608325,0.002279369,0.8766132,0.00068054063,0.000121475954,0.00021388222,0.00048423195,0.0013502054,0.0021737854],"genre_scores_gemma":[0.56726617,0.0012689265,0.42654225,0.00013882475,0.0004897352,0.00021023533,0.001941613,0.00025047362,0.0018918167],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968136,0.0015834243,0.00032057246,0.00059232773,0.0005747551,0.0001153062],"domain_scores_gemma":[0.9846705,0.010240631,0.0018839203,0.0010429437,0.0019325969,0.00022947125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033245555,0.0012263142,0.0009969994,0.004025226,0.00084124925,0.0024901538,0.00075536635,0.0011198987,0.00093759777],"category_scores_gemma":[0.02397137,0.0005320183,0.00077306037,0.0024095858,0.0006215124,0.005029156,0.0013531175,0.0006785095,0.00053316937],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009966069,0.0002481605,0.01733918,0.0027270676,0.00060651044,0.00088239013,0.006635359,0.065304674,0.04592604,0.024228247,0.0073124743,0.82779336],"study_design_scores_gemma":[0.000092751754,0.0007253079,0.020581605,0.00031306554,0.0010919403,0.00094156736,0.0042529064,0.7659903,0.04293973,0.13874528,0.024121353,0.00020413278],"about_ca_topic_score_codex":0.0010799741,"about_ca_topic_score_gemma":0.0015033331,"teacher_disagreement_score":0.004025226,"about_ca_system_score_codex":0.0005924078,"about_ca_system_score_gemma":0.0006183968,"threshold_uncertainty_score":0.017582119},"labels":[],"label_agreement":null},{"id":"W2163284121","doi":"10.1080/07421222.2000.11045646","title":"The Use of Explanations in Knowledge-Based Systems: Cognitive Perspectives and a Process-Tracing Analysis","year":2000,"lang":"en","type":"article","venue":"Journal of Management Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":144,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Protocol analysis; Cognition; Comprehension; Process tracing; Process (computing); Knowledge management; Cognitive psychology; Computer science; Exploratory analysis; Tracing; Psychology; Qualitative analysis; Data science; Qualitative research; Cognitive science","score_opus":0.021402194804987354,"score_gpt":0.289623304492584,"score_spread":0.26822110968759666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2163284121","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7542848,0.0009668987,0.23365855,0.0013713867,0.000017944354,0.00035445768,0.00008353901,0.00018931419,0.009073031],"genre_scores_gemma":[0.96272266,0.00024755704,0.0363978,0.000036062138,0.0000086235805,0.00011293946,0.000050155788,0.000022903525,0.00040134406],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9753984,0.01776651,0.0010680138,0.00095349987,0.0039934632,0.0008200621],"domain_scores_gemma":[0.7954247,0.18621653,0.0076991487,0.0039911224,0.0059692785,0.00069916534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020129886,0.0008722932,0.00044291205,0.0061886273,0.0015458557,0.0058607887,0.0014764239,0.0020033477,0.0011198043],"category_scores_gemma":[0.07860107,0.0006046296,0.0010670268,0.002807876,0.0055405903,0.008002113,0.002582799,0.0015225514,0.00013079536],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006374874,0.0003965838,0.08088252,0.0013577661,0.00019368278,0.0015572284,0.5974468,0.0070627145,0.015071075,0.11101299,0.00043506848,0.18394601],"study_design_scores_gemma":[0.00033460543,0.001571947,0.13841291,0.0036864704,0.0009498251,0.0031792747,0.37036836,0.1299408,0.064273745,0.24834853,0.038378462,0.0005550721],"about_ca_topic_score_codex":0.0025032512,"about_ca_topic_score_gemma":0.0020496715,"teacher_disagreement_score":0.020129886,"about_ca_system_score_codex":0.0025842672,"about_ca_system_score_gemma":0.0019133704,"threshold_uncertainty_score":0.10645831},"labels":[],"label_agreement":null},{"id":"W2163490678","doi":"","title":"Use of Keyphrase Extraction Software for Creation of an AEC/FM Thesaurus","year":2000,"lang":"en","type":"article","venue":"NPARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Thesaurus; Extractor; Computer science; Software; The Internet; Process (computing); Domain (mathematical analysis); Information retrieval; World Wide Web; Software engineering; Natural language processing; Engineering","score_opus":0.02853358991606894,"score_gpt":0.3131675310217468,"score_spread":0.28463394110567786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2163490678","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008685693,0.00022002199,0.9615673,0.00011785062,0.0000890117,0.0010440788,0.0020563465,0.022180397,0.004039361],"genre_scores_gemma":[0.012986402,0.00013805035,0.98097503,0.000031408246,0.000017667477,0.00057223986,0.0021456857,0.0014901,0.001643442],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99826497,0.0003273671,0.0004480761,0.0004450889,0.00045369222,0.000060973252],"domain_scores_gemma":[0.9935941,0.0032723267,0.00043758302,0.00080814795,0.0017583302,0.00012951592],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036192033,0.0016735275,0.001294878,0.0116261635,0.0017194082,0.0026549918,0.0009423865,0.00097378914,0.01154616],"category_scores_gemma":[0.013912078,0.0011210238,0.0015365252,0.007549906,0.0008100573,0.0029398452,0.0016839732,0.0017453548,0.007860834],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002446287,0.000109931854,0.0015388239,0.001948245,0.0001656801,0.00090475136,0.0032103793,0.0017115639,0.06611757,0.010724978,0.011871095,0.9014524],"study_design_scores_gemma":[0.00041103555,0.0006805463,0.011595477,0.00081312115,0.00067482854,0.0063106795,0.002773682,0.12360024,0.30015275,0.029518737,0.52291715,0.00055174483],"about_ca_topic_score_codex":0.0031061203,"about_ca_topic_score_gemma":0.0032595277,"teacher_disagreement_score":0.0116261635,"about_ca_system_score_codex":0.0009826481,"about_ca_system_score_gemma":0.0021018921,"threshold_uncertainty_score":0.038625777},"labels":[],"label_agreement":null},{"id":"W2163630478","doi":"10.1007/978-3-642-04346-8_20","title":"Chance Encounters in the Digital Library","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Novelty; Computer science; Digital library; Interface (matter); Focus (optics); Human–computer interaction; Value (mathematics); Data science; World Wide Web; Machine learning; Psychology","score_opus":0.009265452476631342,"score_gpt":0.23472269564367335,"score_spread":0.225457243167042,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2163630478","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030758297,0.0067923274,0.0009016462,0.012982691,0.0010152613,0.000028912007,0.00013723056,0.0001416775,0.94724196],"genre_scores_gemma":[0.24502061,0.0058851033,0.00064838666,0.0027622909,0.00091230485,0.000036886453,0.0001751804,0.00020939812,0.74434996],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9973043,0.00093179534,0.000091411,0.00024169876,0.00066339935,0.00076735887],"domain_scores_gemma":[0.9966672,0.00059488765,0.00023319453,0.00009620531,0.00015554043,0.002252939],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013945583,0.0004455944,0.0007346394,0.0020434686,0.021584984,0.021630954,0.0012395391,0.0029226104,0.13566124],"category_scores_gemma":[0.0048526344,0.00057280913,0.00055459223,0.003126315,0.006196739,0.016828874,0.012882307,0.004588015,0.022456069],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023304234,0.00019420419,0.0067413687,0.00033853954,0.000019976767,0.0023001123,0.13524522,0.00012523102,0.00043160334,0.34613034,0.39603493,0.11220549],"study_design_scores_gemma":[0.0000072936637,0.00003546597,0.0017150249,0.00025452496,0.0000059603904,0.0011474063,0.07074772,0.00006140039,0.00010723747,0.012742201,0.9131522,0.000023555374],"about_ca_topic_score_codex":0.0062264623,"about_ca_topic_score_gemma":0.026956566,"teacher_disagreement_score":0.13566124,"about_ca_system_score_codex":0.0043373443,"about_ca_system_score_gemma":0.00444168,"threshold_uncertainty_score":0.4538321},"labels":[],"label_agreement":null},{"id":"W2169142063","doi":"10.1613/jair.3940","title":"Topic Segmentation and Labeling in Asynchronous Conversations","year":2013,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; University of British Columbia","keywords":"Computer science; Asynchronous communication; Segmentation; Artificial intelligence; Conversation; Natural language processing; Exploit; Graph; Linguistics; Theoretical computer science","score_opus":0.1082659095785678,"score_gpt":0.424850168063926,"score_spread":0.3165842584853582,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169142063","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20048682,0.0009937906,0.7877063,0.0006492131,0.00017709467,0.0003308395,0.0013712787,0.0027800666,0.0055046426],"genre_scores_gemma":[0.77800834,0.00037590263,0.2129153,0.00018747503,0.00025954362,0.00052520557,0.0035939857,0.0005529955,0.0035813113],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99184847,0.0045374283,0.00035620743,0.0020923137,0.00084014464,0.0003254047],"domain_scores_gemma":[0.9691306,0.022606453,0.001983828,0.0029485552,0.0026201531,0.0007104086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005408508,0.0010153779,0.0010349472,0.0024022565,0.00209882,0.0025906332,0.0016323879,0.0017663252,0.0018189559],"category_scores_gemma":[0.029277477,0.0006232549,0.0009398351,0.0018323228,0.00122094,0.0038674322,0.0023274356,0.0017402727,0.0015505193],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0049327784,0.00070071244,0.053772256,0.001993169,0.0004754022,0.0014616782,0.022759428,0.16416532,0.09098533,0.05478608,0.026906628,0.57706124],"study_design_scores_gemma":[0.00009919636,0.0001399109,0.012678887,0.00009963029,0.0001345451,0.00033942674,0.0016913769,0.8910539,0.023743302,0.052916232,0.016974473,0.00012922268],"about_ca_topic_score_codex":0.0052136322,"about_ca_topic_score_gemma":0.007229337,"teacher_disagreement_score":0.005408508,"about_ca_system_score_codex":0.0012299822,"about_ca_system_score_gemma":0.00148746,"threshold_uncertainty_score":0.028603256},"labels":[],"label_agreement":null},{"id":"W2169517221","doi":"10.1177/0146621610391777","title":"Accuracy of Person-Fit Statistics","year":2011,"lang":"en","type":"article","venue":"Applied Psychological Measurement","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; Université de Sherbrooke","funders":"","keywords":"Statistics; Cheating; Monte Carlo method; Mathematics; Econometrics; Psychology; Social psychology","score_opus":0.24308051572562275,"score_gpt":0.34340061958845114,"score_spread":0.10032010386282839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169517221","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8019421,0.0004197682,0.17949001,0.0003535234,0.00014613246,0.00070556795,0.0007901518,0.0010858861,0.015066776],"genre_scores_gemma":[0.98561096,0.00005758883,0.012626367,0.00007410775,0.000020130046,0.00019817139,0.00031221868,0.00015910843,0.00094135443],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9520745,0.024668708,0.004808754,0.008370388,0.008795697,0.0012819505],"domain_scores_gemma":[0.4484861,0.4514304,0.03982987,0.043614548,0.015100657,0.0015384483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04380894,0.00074738456,0.0011242764,0.0018077277,0.00045115792,0.002905105,0.0009182058,0.0017810024,0.0059001204],"category_scores_gemma":[0.4940274,0.00049063633,0.0011334367,0.0013780922,0.0017011798,0.0057670665,0.0025837573,0.0017218024,0.0010817961],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.015110283,0.0014353876,0.44398,0.0010112957,0.0017262072,0.00037283087,0.011438966,0.064362414,0.03489734,0.040260345,0.0034486267,0.38195625],"study_design_scores_gemma":[0.00047567228,0.007966646,0.6039726,0.00029773198,0.00075702835,0.0014002187,0.0014438964,0.24675654,0.06868179,0.057924367,0.009689617,0.0006338442],"about_ca_topic_score_codex":0.0013354752,"about_ca_topic_score_gemma":0.00061869127,"teacher_disagreement_score":0.04380894,"about_ca_system_score_codex":0.00088933145,"about_ca_system_score_gemma":0.0006506518,"threshold_uncertainty_score":0.23168659},"labels":[],"label_agreement":null},{"id":"W2170783728","doi":"10.1002/meet.14505001072","title":"Exact versus estimated pruning of subject hierarchies","year":2013,"lang":"en","type":"article","venue":"Proceedings of the American Society for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Subject (documents); Hierarchy; Computer science; Pruning; Data science; Information retrieval; Visualization; Artificial intelligence; Theoretical computer science; World Wide Web","score_opus":0.013886144264422492,"score_gpt":0.2833045487715939,"score_spread":0.2694184045071714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170783728","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.71583486,0.0019347095,0.27384642,0.00031158858,0.000089468245,0.00025261793,0.00084115035,0.0026239953,0.0042651827],"genre_scores_gemma":[0.7762066,0.0004017142,0.22096097,0.000050011382,0.00005036797,0.00009072471,0.0012015284,0.00020664874,0.0008314025],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967894,0.0013621408,0.00021835338,0.00054687844,0.0008547446,0.00022847303],"domain_scores_gemma":[0.97540754,0.017795146,0.0016113488,0.0029966326,0.0018476655,0.0003416594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030266936,0.00052160345,0.00073889835,0.0020598352,0.0006048754,0.002131504,0.0009527929,0.0006196313,0.001463126],"category_scores_gemma":[0.030226843,0.0002837359,0.00042154657,0.0015044457,0.00054269994,0.0022832137,0.0009492542,0.0007418052,0.0004425025],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002237882,0.00058494334,0.051384773,0.00080628076,0.0002541497,0.00043319585,0.0020512675,0.14640017,0.04394684,0.015167936,0.0069538374,0.7297787],"study_design_scores_gemma":[0.000111803405,0.00060508226,0.031310085,0.00019322464,0.0001666298,0.0006429885,0.0008878756,0.9244447,0.02428992,0.009760058,0.007533761,0.00005389779],"about_ca_topic_score_codex":0.0037125577,"about_ca_topic_score_gemma":0.005276326,"teacher_disagreement_score":0.0037125577,"about_ca_system_score_codex":0.00050762494,"about_ca_system_score_gemma":0.0011321473,"threshold_uncertainty_score":0.016006887},"labels":[],"label_agreement":null},{"id":"W2177607365","doi":"10.1007/s10515-015-0184-4","title":"Concept extraction from business documents for software engineering projects","year":2015,"lang":"en","type":"article","venue":"Automated Software Engineering","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Documentation; Domain engineering; Software mining; Software engineering; Process (computing); Domain (mathematical analysis); Software; Data science; Knowledge extraction; Data mining; Domain analysis; Information extraction; Software development; Information retrieval; Software construction; Programming language","score_opus":0.016569986768452373,"score_gpt":0.2722598448418184,"score_spread":0.255689858073366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2177607365","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28676787,0.011577192,0.5887855,0.0032806585,0.0008062004,0.0030749415,0.0692646,0.015674269,0.020768726],"genre_scores_gemma":[0.25721994,0.0032352593,0.67334247,0.000252925,0.0002212796,0.00096203486,0.058846865,0.00077651715,0.005142631],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997726,0.00041773485,0.00037886217,0.00038342862,0.00094693096,0.00014705407],"domain_scores_gemma":[0.98923254,0.006608885,0.0010284492,0.00057325937,0.0022441144,0.00031271242],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001706957,0.0012460605,0.0007174063,0.012921057,0.0015247915,0.0023807145,0.0009696032,0.0013353595,0.0037707046],"category_scores_gemma":[0.011981763,0.0005525166,0.0010968826,0.008538714,0.00046541975,0.003390618,0.0011940093,0.00153809,0.003378985],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049901014,0.00045445008,0.011508349,0.0042537507,0.00015261368,0.0016701482,0.002118509,0.0031998232,0.0776017,0.010008922,0.044201694,0.844331],"study_design_scores_gemma":[0.00039844017,0.00091674284,0.06676876,0.002857126,0.0014708269,0.007678938,0.0057836254,0.12820683,0.23392718,0.042281173,0.5093444,0.0003659638],"about_ca_topic_score_codex":0.0033277487,"about_ca_topic_score_gemma":0.00469067,"teacher_disagreement_score":0.012921057,"about_ca_system_score_codex":0.0010533625,"about_ca_system_score_gemma":0.004368753,"threshold_uncertainty_score":0.01261425},"labels":[],"label_agreement":null},{"id":"W2184546122","doi":"","title":"PRIS at Knowledge Base Population 2013","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Bootstrapping (finance); Computer science; Ranking (information retrieval); Task (project management); Knowledge base; Entity linking; Population; Similarity (geometry); Base (topology); Artificial intelligence; Data mining; Information retrieval; Natural language processing; Machine learning; Mathematics; Image (mathematics); Engineering","score_opus":0.007412972671835042,"score_gpt":0.2606132115266101,"score_spread":0.25320023885477505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2184546122","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018202635,0.0005022482,0.94214547,0.00092075823,0.00030357504,0.001393432,0.00369749,0.014266851,0.01856755],"genre_scores_gemma":[0.16820435,0.00038408037,0.77870363,0.00078670046,0.00025923856,0.0017343818,0.020585803,0.0019128139,0.027429052],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996351,0.0006651197,0.00022564434,0.0011891387,0.0012017076,0.00036734203],"domain_scores_gemma":[0.9961817,0.0010921549,0.00019379615,0.001242293,0.0010399197,0.00025015837],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002983205,0.0011617518,0.0015344773,0.005662848,0.00229669,0.0043036933,0.003615455,0.0017813644,0.019197151],"category_scores_gemma":[0.014097984,0.0009823247,0.0020887253,0.0049365866,0.0007713851,0.005271608,0.0050316406,0.003461857,0.009132267],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037575813,0.0003705805,0.003532964,0.00028377943,0.00017234284,0.00039703984,0.00049350376,0.015583286,0.0056782933,0.04095159,0.05923098,0.8729299],"study_design_scores_gemma":[0.00023451819,0.00035859985,0.00531552,0.00026028033,0.00037045262,0.0013036607,0.0010663151,0.52006257,0.0246582,0.19764249,0.24852303,0.00020446556],"about_ca_topic_score_codex":0.0055601425,"about_ca_topic_score_gemma":0.007623017,"teacher_disagreement_score":0.019197151,"about_ca_system_score_codex":0.0018467543,"about_ca_system_score_gemma":0.0035580464,"threshold_uncertainty_score":0.064220846},"labels":[],"label_agreement":null},{"id":"W2207763101","doi":"","title":"Social Annotation: Emergent Text Signals through Self-Organization","year":2010,"lang":"en","type":"article","venue":"EdMedia: World Conference on Educational Media and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Athabasca University","funders":"","keywords":"Annotation; Computer science; Natural language processing; Artificial intelligence; Data science","score_opus":0.020922287943656633,"score_gpt":0.29768968425692305,"score_spread":0.2767673963132664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2207763101","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.50528073,0.00087611814,0.44771037,0.0030530193,0.0007643154,0.0002573889,0.0015913195,0.0034659372,0.037000842],"genre_scores_gemma":[0.95427024,0.00023196891,0.037789397,0.00018485129,0.00037369557,0.00012601724,0.0007651967,0.00043302568,0.0058256197],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99883515,0.00035648656,0.00005605479,0.00032067104,0.0003517475,0.000079940255],"domain_scores_gemma":[0.9881166,0.0073398342,0.0012103522,0.0013418937,0.0015514322,0.0004398473],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012320926,0.0003769802,0.0003126957,0.0018075024,0.0010505599,0.0027737366,0.000698969,0.000974372,0.0038069745],"category_scores_gemma":[0.015603965,0.00037188886,0.00024036656,0.0019563641,0.001304337,0.004714524,0.0020556073,0.0010401104,0.0012992498],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012359277,0.0004054382,0.035239935,0.000963165,0.00013464697,0.0018267762,0.021834873,0.010751909,0.24336618,0.12715305,0.025157297,0.53193086],"study_design_scores_gemma":[0.0001019076,0.00041160244,0.07372777,0.00020064702,0.0001877375,0.0010527148,0.008123277,0.5100941,0.092197716,0.25789255,0.055795573,0.00021443689],"about_ca_topic_score_codex":0.0006499614,"about_ca_topic_score_gemma":0.00082952954,"teacher_disagreement_score":0.0038069745,"about_ca_system_score_codex":0.00045583508,"about_ca_system_score_gemma":0.00044571763,"threshold_uncertainty_score":0.012735605},"labels":[],"label_agreement":null},{"id":"W2212854643","doi":"10.15837/ijccc.2011.3.2132","title":"Human-inspired Identification of High-level Concepts using OWA and Linguistic Quantifiers","year":2011,"lang":"en","type":"article","venue":"International Journal of Computers Communications & Control","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Schema (genetic algorithms); Schema matching; Identification (biology); Matching (statistics); Artificial intelligence; Natural language processing; Machine learning; Data mining; Mathematics; Data integration","score_opus":0.08137653461828818,"score_gpt":0.3690841034683447,"score_spread":0.2877075688500565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2212854643","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057093505,0.00027288691,0.9402336,0.00015935444,0.000021559843,0.00016247449,0.000057056,0.00039664775,0.0016029358],"genre_scores_gemma":[0.37932068,0.00010752845,0.6188943,0.00004538354,0.000018392546,0.00017278388,0.000081848906,0.00004099286,0.001318136],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99862087,0.0004774933,0.00013437576,0.00027608033,0.00043194197,0.000059246267],"domain_scores_gemma":[0.9972372,0.0016385261,0.00044269284,0.00014514488,0.0004452455,0.00009118961],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028215435,0.00051931205,0.0005757065,0.0025062272,0.00066176965,0.0016008553,0.000997417,0.00053967815,0.0013001801],"category_scores_gemma":[0.0072536417,0.0002772004,0.00069630303,0.0013583396,0.001180698,0.0032585738,0.0009889633,0.00067265757,0.00015467426],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047940793,0.00038110756,0.011905515,0.0010847667,0.0003781273,0.0007399604,0.0055368044,0.07181891,0.09068895,0.18961866,0.0021563536,0.62521136],"study_design_scores_gemma":[0.000052615786,0.00020058703,0.005984963,0.00008315453,0.00015637951,0.00032152914,0.0009086261,0.84410655,0.028337326,0.11229822,0.007453527,0.00009645662],"about_ca_topic_score_codex":0.0037731526,"about_ca_topic_score_gemma":0.0040899264,"teacher_disagreement_score":0.0037731526,"about_ca_system_score_codex":0.0011198107,"about_ca_system_score_gemma":0.0010231131,"threshold_uncertainty_score":0.014921963},"labels":[],"label_agreement":null},{"id":"W2219965243","doi":"","title":"The RhetFig project: computational rhetorics and models of persuasion","year":2011,"lang":"en","type":"article","venue":"National Conference on Artificial Intelligence","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Persuasion; Automatic summarization; Topos theory; Ontology; Computer science; Rhetoric; Annotation; Ontology engineering; Natural language processing; Artificial intelligence; Epistemology; Linguistics; Process ontology; Semantic Web; Philosophy","score_opus":0.26881621305356274,"score_gpt":0.3706189476740427,"score_spread":0.10180273462047995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2219965243","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036702953,0.0074837077,0.8523303,0.018176274,0.0008543186,0.0002658392,0.0024792138,0.0030650839,0.078642234],"genre_scores_gemma":[0.5511858,0.004320192,0.41509545,0.0014902742,0.00077316107,0.0010365815,0.004077108,0.001201766,0.020819666],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974775,0.0016971043,0.000069655594,0.00039322942,0.0002708248,0.0000916883],"domain_scores_gemma":[0.99299663,0.005136868,0.00028172636,0.000933362,0.00038532444,0.00026609312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046970253,0.0010409604,0.0011761452,0.0023672904,0.001538755,0.006246943,0.0025421388,0.0030243609,0.016980374],"category_scores_gemma":[0.013074611,0.0006166809,0.0016072914,0.0019241477,0.003534388,0.009711933,0.0029443197,0.003014505,0.0022362226],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057365614,0.00005588864,0.00047662674,0.00016917597,0.000055416545,0.00006409845,0.0009304714,0.012554695,0.00016041855,0.94926745,0.010208009,0.0260004],"study_design_scores_gemma":[0.000038415263,0.00001640463,0.00021950703,0.00008341413,0.000016889782,0.00003996898,0.00018482721,0.049785078,0.0001582439,0.92634183,0.023101708,0.000013626934],"about_ca_topic_score_codex":0.0044822656,"about_ca_topic_score_gemma":0.002460362,"teacher_disagreement_score":0.016980374,"about_ca_system_score_codex":0.0021717409,"about_ca_system_score_gemma":0.0017176935,"threshold_uncertainty_score":0.056805015},"labels":[],"label_agreement":null},{"id":"W2230887875","doi":"10.1080/10888438.2015.1107073","title":"The Random Forests statistical technique: An examination of its value for the study of reading","year":2016,"lang":"en","type":"article","venue":"Scientific Studies of Reading","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":168,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"Eunice Kennedy Shriver National Institute of Child Health and Human Development; National Institute of Child Health and Human Development; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Reading (process); Computer science; Random forest; Statistical analysis; Value (mathematics); Statistics; Artificial intelligence; Machine learning; Mathematics; Linguistics","score_opus":0.047122001145699366,"score_gpt":0.36743177930217613,"score_spread":0.32030977815647677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2230887875","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03267493,0.0064652544,0.9527307,0.002163862,0.00033816986,0.0003618429,0.00046085077,0.0008896645,0.003914651],"genre_scores_gemma":[0.29945096,0.0035187674,0.69331974,0.00061108003,0.00042845326,0.00071296946,0.00036839437,0.00060338096,0.000986213],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.95467687,0.03651902,0.0011135343,0.0020630497,0.0052690334,0.0003585364],"domain_scores_gemma":[0.57903075,0.39933565,0.005000676,0.008267893,0.007350882,0.0010142046],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.11568536,0.0014684917,0.0021628353,0.0056741275,0.0015031119,0.0024906062,0.0017034351,0.0018282903,0.0024391704],"category_scores_gemma":[0.24447149,0.00056472665,0.001884998,0.0064348243,0.002702569,0.0037233534,0.0017966289,0.003559293,0.0006430743],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083113654,0.0002468865,0.082586855,0.0016878706,0.002353222,0.0006815543,0.0036805286,0.061752636,0.0026236197,0.09100565,0.01331787,0.7392322],"study_design_scores_gemma":[0.00027104863,0.0012605924,0.06560376,0.0027148938,0.00091417914,0.0024571456,0.002110597,0.51264286,0.0036543373,0.366918,0.04095934,0.0004932142],"about_ca_topic_score_codex":0.0056409296,"about_ca_topic_score_gemma":0.0072705173,"teacher_disagreement_score":0.88431466,"about_ca_system_score_codex":0.00089148304,"about_ca_system_score_gemma":0.0028119232,"threshold_uncertainty_score":0.61180997},"labels":[],"label_agreement":null},{"id":"W2250467175","doi":"","title":"An initial study of topical poetry segmentation","year":2013,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Poetry; Focus (optics); Segmentation; Test (biology); Computer science; Artificial intelligence; Natural language processing; Art; Literature; Geology","score_opus":0.018801084788205018,"score_gpt":0.3477503906298853,"score_spread":0.32894930584168025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250467175","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.873304,0.0005401445,0.075257555,0.0011244451,0.00017147759,0.0024057576,0.00088544807,0.00034537067,0.04596583],"genre_scores_gemma":[0.9528524,0.00030088765,0.038625386,0.0002593423,0.00007114068,0.0015444835,0.0009093641,0.000302248,0.005134841],"study_design_codex":"qualitative","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9745306,0.017112434,0.0014288982,0.0022865806,0.003924316,0.000717176],"domain_scores_gemma":[0.79463595,0.13666007,0.00718186,0.014207791,0.045541983,0.0017722676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023903543,0.00040914892,0.0006811413,0.0034056748,0.004766664,0.0033654144,0.0014690387,0.0008246943,0.0067212256],"category_scores_gemma":[0.16681232,0.0005569649,0.00035884333,0.0042395256,0.005189792,0.006467757,0.0044133808,0.0027179592,0.0016684026],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006447887,0.00043030418,0.031176608,0.0017887182,0.000025645539,0.00044919422,0.7764717,0.00036127138,0.03633002,0.014648408,0.0030935043,0.13457982],"study_design_scores_gemma":[0.000119238925,0.002233928,0.16674347,0.001601703,0.00008714318,0.0016100296,0.64534485,0.006112362,0.047269396,0.015304998,0.113369726,0.00020310983],"about_ca_topic_score_codex":0.0030427827,"about_ca_topic_score_gemma":0.0047463216,"teacher_disagreement_score":0.023903543,"about_ca_system_score_codex":0.002288329,"about_ca_system_score_gemma":0.0025595096,"threshold_uncertainty_score":0.12641555},"labels":[],"label_agreement":null},{"id":"W2250595592","doi":"10.3115/v1/w14-4407","title":"A Template-based Abstractive Meeting Summarization: Leveraging Summary and Source Text Relationships","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":100,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Automatic summarization; Readability; Computer science; Template; Sentence; Natural language processing; Information retrieval; Artificial intelligence; Programming language","score_opus":0.02111276075879438,"score_gpt":0.24509023218057135,"score_spread":0.22397747142177699,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250595592","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013027696,0.00085592415,0.9563157,0.00037809633,0.00029428964,0.0004091952,0.0026392494,0.023982603,0.0020973133],"genre_scores_gemma":[0.061673906,0.00048430334,0.9239806,0.00013535586,0.00030096815,0.0002989705,0.0085094,0.0009308405,0.0036857142],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99748355,0.00071096857,0.00033448284,0.000733846,0.0006602872,0.00007688653],"domain_scores_gemma":[0.99416035,0.0019233179,0.00092092,0.00085458707,0.0019217146,0.00021906439],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021172834,0.0018280796,0.0009840046,0.003078339,0.00080851454,0.002457667,0.0017081529,0.0012744743,0.0059930026],"category_scores_gemma":[0.009471575,0.00057445595,0.000982159,0.0017299775,0.00040950032,0.002700882,0.0013336989,0.0013049416,0.0074856034],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045214276,0.00015143269,0.001335132,0.0012491267,0.00019241152,0.00032547265,0.001098783,0.006429773,0.13613227,0.0023807273,0.023863375,0.8263894],"study_design_scores_gemma":[0.00025208798,0.0014074614,0.009510899,0.00032689914,0.0010127124,0.001918121,0.0014571652,0.4429461,0.36574337,0.010349875,0.16461025,0.0004650438],"about_ca_topic_score_codex":0.0015917396,"about_ca_topic_score_gemma":0.0019332865,"teacher_disagreement_score":0.0059930026,"about_ca_system_score_codex":0.0003801481,"about_ca_system_score_gemma":0.00090124237,"threshold_uncertainty_score":0.020048618},"labels":[],"label_agreement":null},{"id":"W2251442452","doi":"10.3115/v1/p14-1115","title":"Abstractive Summarization of Spoken and Written Conversations Based on Phrasal Queries","year":2014,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Automatic summarization; Computer science; Natural language processing; Multi-document summarization; Artificial intelligence; Information retrieval","score_opus":0.005305981004063103,"score_gpt":0.23099926973793516,"score_spread":0.22569328873387207,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251442452","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041758712,0.0011956092,0.9394748,0.0004993344,0.00019381878,0.00040304757,0.0023362346,0.0119619835,0.0021764815],"genre_scores_gemma":[0.2611777,0.0011046175,0.7110545,0.000334409,0.00060121255,0.00056296395,0.016171077,0.0013804013,0.0076132086],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99826694,0.00051198265,0.00017786307,0.00043177075,0.00050267455,0.000108687884],"domain_scores_gemma":[0.9960967,0.0013901675,0.0004944618,0.00051788526,0.0013618629,0.00013893808],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012281272,0.0020450056,0.0014499114,0.0026898154,0.00081588514,0.0019144698,0.0013710244,0.000820724,0.0027165108],"category_scores_gemma":[0.0053407145,0.0005118872,0.001130927,0.0017431232,0.0004414082,0.0028181337,0.0014508879,0.0012709583,0.0027881889],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010411878,0.0002678114,0.002093774,0.0013215296,0.00028784576,0.000550044,0.0029885375,0.022316903,0.18324195,0.007204034,0.02664908,0.75203735],"study_design_scores_gemma":[0.00013136017,0.0008437045,0.007159353,0.00013663557,0.0007221958,0.00063369726,0.00206622,0.7779591,0.12593256,0.025803851,0.05841325,0.00019803607],"about_ca_topic_score_codex":0.0034978346,"about_ca_topic_score_gemma":0.004452844,"teacher_disagreement_score":0.0034978346,"about_ca_system_score_codex":0.00056625984,"about_ca_system_score_gemma":0.000986643,"threshold_uncertainty_score":0.009087622},"labels":[],"label_agreement":null},{"id":"W2252024428","doi":"","title":"Towards Topic Labeling with Phrase Entailment and Aggregation","year":2013,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Textual entailment; Phrase; Computer science; Logical consequence; Natural language processing; Artificial intelligence; Generalization; Set (abstract data type); Aggregate (composite); Graph; Mathematics; Theoretical computer science","score_opus":0.006640097068752401,"score_gpt":0.23190119143036342,"score_spread":0.22526109436161101,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252024428","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0035343273,0.00040146167,0.99281704,0.0002230951,0.00003998986,0.000092767616,0.00036237165,0.002005347,0.0005235309],"genre_scores_gemma":[0.04552309,0.00034477518,0.94934165,0.00024631113,0.00027603196,0.00027227105,0.0026421887,0.00043227733,0.0009212818],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99218786,0.0034332117,0.00043433858,0.0020450167,0.0015356338,0.0003639989],"domain_scores_gemma":[0.9883732,0.006209247,0.00095894065,0.002392918,0.0017561903,0.0003095122],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008288027,0.002715684,0.0026690846,0.0071826265,0.0023773308,0.00427805,0.0037915152,0.0029637557,0.0024359133],"category_scores_gemma":[0.018802967,0.0013892779,0.0025466199,0.008808436,0.0017217835,0.008492661,0.0053785085,0.004713034,0.0023800374],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007606822,0.0005387238,0.0062672016,0.00081285153,0.00052703655,0.00038736008,0.0025365022,0.07220546,0.025505044,0.07845842,0.037144374,0.77485627],"study_design_scores_gemma":[0.00007996592,0.000092918366,0.0010496221,0.000077185505,0.00017010691,0.00020437487,0.0002892613,0.7961249,0.012419157,0.17325458,0.016163172,0.000074755764],"about_ca_topic_score_codex":0.009747291,"about_ca_topic_score_gemma":0.011909695,"teacher_disagreement_score":0.009747291,"about_ca_system_score_codex":0.0022007267,"about_ca_system_score_gemma":0.00406756,"threshold_uncertainty_score":0.043831766},"labels":[],"label_agreement":null},{"id":"W2252545675","doi":"10.1007/978-3-319-25252-0_18","title":"Ontology-Based Topic Labeling and Quality Prediction","year":2015,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Latent Dirichlet allocation; Topic model; Reliability (semiconductor); Probabilistic logic; Ontology; Coherence (philosophical gambling strategy); Quality (philosophy); Artificial intelligence; Information retrieval; Natural language processing; Semantics (computer science); Machine learning; Data mining","score_opus":0.0454640940380162,"score_gpt":0.31517178953091995,"score_spread":0.26970769549290374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252545675","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043418653,0.0031246326,0.92805415,0.0006504987,0.0003480347,0.00041182875,0.0071149617,0.008203916,0.008673312],"genre_scores_gemma":[0.29016024,0.0020346693,0.67648613,0.000138092,0.00035242853,0.00043163693,0.020762416,0.001153326,0.008481096],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9969625,0.0004640749,0.00027009987,0.00072423345,0.0013169841,0.0002621508],"domain_scores_gemma":[0.99369276,0.002568817,0.0004890771,0.0007708321,0.002213885,0.00026470507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026688285,0.0010772331,0.0011443812,0.0078649735,0.0011903311,0.003012697,0.0018687361,0.001287579,0.0046787793],"category_scores_gemma":[0.011193294,0.00054934743,0.0017898167,0.007539602,0.0005827153,0.0048845364,0.0021039986,0.0017584714,0.00344136],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005502708,0.0003872318,0.013971468,0.00089236966,0.00021865059,0.00022115816,0.00076690246,0.017948804,0.030087305,0.013892931,0.03521571,0.8858472],"study_design_scores_gemma":[0.00010979729,0.0001598788,0.018810201,0.00022491782,0.00048309017,0.0005512571,0.0007067506,0.8432064,0.03183243,0.051619284,0.052149218,0.00014676734],"about_ca_topic_score_codex":0.011201128,"about_ca_topic_score_gemma":0.01596886,"teacher_disagreement_score":0.011201128,"about_ca_system_score_codex":0.0016841976,"about_ca_system_score_gemma":0.0026638475,"threshold_uncertainty_score":0.022271872},"labels":[],"label_agreement":null},{"id":"W2271445843","doi":"","title":"Interpersonal Awareness During Web-based Concept Mapping: The Effect of Different Communication Channels","year":2004,"lang":"en","type":"article","venue":"EdMedia: World Conference on Educational Media and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Interpersonal communication; Computer science; Interpersonal relationship; World Wide Web; Psychology; Knowledge management; Internet privacy; Social psychology; Communication","score_opus":0.021766113883649373,"score_gpt":0.2753001712164939,"score_spread":0.25353405733284456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2271445843","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9878028,0.00020495385,0.006620138,0.00010461092,0.000047763093,0.00008194265,0.00005275017,0.00023190516,0.0048531457],"genre_scores_gemma":[0.9953245,0.00009615381,0.003412899,0.00006436245,0.000017843786,0.00006980277,0.000079297206,0.00010954866,0.00082568376],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99713683,0.0013761895,0.00016258236,0.00035616956,0.00075049885,0.00021766324],"domain_scores_gemma":[0.90636986,0.0836937,0.0034799865,0.0023512747,0.0027420616,0.0013631242],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028409692,0.00037323232,0.00041290477,0.00089870836,0.00056645443,0.0029196064,0.0006136619,0.00094475906,0.0034199776],"category_scores_gemma":[0.08616646,0.00051728747,0.00027108533,0.00057775545,0.00045655062,0.0032684866,0.0019564526,0.0013767135,0.00039600756],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013881765,0.0033086706,0.09250512,0.0020692658,0.0006121687,0.0019771962,0.145619,0.0052500693,0.3299235,0.003252957,0.003551616,0.39804867],"study_design_scores_gemma":[0.00070273055,0.0052331733,0.7243991,0.0007622133,0.0021405353,0.002545364,0.06911378,0.062234864,0.10946299,0.01342202,0.009432769,0.00055053184],"about_ca_topic_score_codex":0.0015245075,"about_ca_topic_score_gemma":0.0013469164,"teacher_disagreement_score":0.0034199776,"about_ca_system_score_codex":0.00028683874,"about_ca_system_score_gemma":0.00060914614,"threshold_uncertainty_score":0.015024662},"labels":[],"label_agreement":null},{"id":"W2278408372","doi":"10.1007/3-540-45153-6_27","title":"Évaluation d’un Système pour le Résumé Automatique de Documents ÉLectroniques","year":2001,"lang":"fr","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Valuation (finance); Business; Accounting","score_opus":0.024545233327482822,"score_gpt":0.2934102087685686,"score_spread":0.2688649754410858,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2278408372","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30024704,0.0030248468,0.5664443,0.0015264841,0.0009202599,0.0015696274,0.0032259908,0.112072185,0.01096941],"genre_scores_gemma":[0.4645972,0.0009357813,0.49544632,0.00041935343,0.0002296612,0.0006681116,0.0059283413,0.0022389642,0.029536227],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99588674,0.0010538755,0.00039771234,0.000901599,0.0016086849,0.00015141451],"domain_scores_gemma":[0.9883145,0.0072901226,0.00028070563,0.0012521978,0.0025836385,0.00027884886],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036955453,0.0012709746,0.0011816933,0.0018131005,0.00088888715,0.003468412,0.0014662071,0.0021424785,0.011385343],"category_scores_gemma":[0.012889998,0.0008071897,0.00087449065,0.0008698145,0.00074250076,0.0029717798,0.0009936747,0.0010220094,0.0045220694],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0056735156,0.0011731925,0.0056854943,0.0015093649,0.00031546445,0.00061406376,0.0011717227,0.020851301,0.15375105,0.004055008,0.018847814,0.7863519],"study_design_scores_gemma":[0.0014753992,0.0021611718,0.014701883,0.00023152547,0.00046079876,0.0012178198,0.0008951877,0.5079019,0.3748596,0.0022305085,0.093665265,0.00019899596],"about_ca_topic_score_codex":0.015024905,"about_ca_topic_score_gemma":0.0072352192,"teacher_disagreement_score":0.015024905,"about_ca_system_score_codex":0.0010132316,"about_ca_system_score_gemma":0.001695319,"threshold_uncertainty_score":0.038087785},"labels":[],"label_agreement":null},{"id":"W2294408224","doi":"10.1007/s12080-016-0292-1","title":"Introduction to the special issue: theory of food webs","year":2016,"lang":"en","type":"article","venue":"Theoretical Ecology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Computer science; Mathematical economics; Mathematics","score_opus":0.004947937222468709,"score_gpt":0.24462835334006192,"score_spread":0.23968041611759322,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2294408224","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00059758953,0.061829112,0.043369345,0.07362317,0.7845469,0.00007070715,0.00092484656,0.00067160866,0.034366604],"genre_scores_gemma":[0.004803976,0.036215153,0.0113667175,0.018807597,0.8569268,0.0001409657,0.001439318,0.00082570314,0.06947373],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983981,0.00035704186,0.00019000402,0.00042258162,0.0005358883,0.00009631654],"domain_scores_gemma":[0.98780847,0.0068674814,0.00048697033,0.000893988,0.0024472915,0.0014958251],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025981178,0.0020795846,0.002801216,0.0051984303,0.0015052754,0.0065692854,0.0023121412,0.0039034325,0.061294317],"category_scores_gemma":[0.010422045,0.0007056237,0.0028147015,0.0037298028,0.0020541805,0.006440393,0.0025025597,0.008305223,0.028896453],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016855642,0.00004323183,0.00013862435,0.00031685558,0.00003086565,0.00005779174,0.0000335093,0.00024168192,0.00026703192,0.017543651,0.9332504,0.048059512],"study_design_scores_gemma":[0.00001592653,0.00005848684,0.0007145062,0.00035321477,0.0000407835,0.00036620782,0.000055297965,0.001504601,0.00018357835,0.06746164,0.92920035,0.000045530876],"about_ca_topic_score_codex":0.0005669365,"about_ca_topic_score_gemma":0.0011322119,"teacher_disagreement_score":0.061294317,"about_ca_system_score_codex":0.0014525037,"about_ca_system_score_gemma":0.0016792726,"threshold_uncertainty_score":0.20504993},"labels":[],"label_agreement":null},{"id":"W2298270464","doi":"","title":"The distribution of references in scientific papers: An analysis of the IMRaD structure","year":2013,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Categorization; Section (typography); Matching (statistics); Information retrieval; Scientific literature; Data science; Artificial intelligence; Mathematics; Statistics","score_opus":0.01170373291071532,"score_gpt":0.24597000095097465,"score_spread":0.23426626804025932,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2298270464","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44799328,0.006961156,0.38570157,0.009344597,0.0012885727,0.0003777831,0.010162325,0.0030248023,0.13514595],"genre_scores_gemma":[0.91907054,0.0028962751,0.044937816,0.0003710741,0.0014903019,0.00030470497,0.0044208607,0.0009965616,0.02551193],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9948243,0.0012089252,0.0005245713,0.0009217732,0.0019201695,0.0006004073],"domain_scores_gemma":[0.8996025,0.05146742,0.015047755,0.016159367,0.013626316,0.004096658],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.008589394,0.0005391054,0.0013012548,0.016687214,0.0022929802,0.00552037,0.0031371028,0.0015843328,0.024598101],"category_scores_gemma":[0.08535199,0.0006617469,0.0010945988,0.016937679,0.002876858,0.008900995,0.0035679657,0.001962329,0.0051054102],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001106191,0.00022038387,0.028846819,0.00055950746,0.00016571066,0.0006403144,0.0013201851,0.015628377,0.0063871653,0.8077758,0.01725004,0.12009935],"study_design_scores_gemma":[0.00019166012,0.00039480545,0.037852556,0.0004201428,0.00031167117,0.0016885073,0.0010680527,0.16693252,0.011117452,0.7292409,0.050598443,0.00018328583],"about_ca_topic_score_codex":0.0007662015,"about_ca_topic_score_gemma":0.0006384489,"teacher_disagreement_score":0.9914106,"about_ca_system_score_codex":0.0020650916,"about_ca_system_score_gemma":0.0017934273,"threshold_uncertainty_score":0.0822888},"labels":[],"label_agreement":null},{"id":"W2302088940","doi":"10.7202/1029091ar","title":"Vers une nouvelle génération d’outils d’analyse et de recherche d’information","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Political science; Philosophy","score_opus":0.11358409482440322,"score_gpt":0.41908623043501403,"score_spread":0.3055021356106108,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2302088940","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02288366,0.014325299,0.9068884,0.013886358,0.0010361578,0.00036331426,0.0011001437,0.003685791,0.035830945],"genre_scores_gemma":[0.11371451,0.011593143,0.8378947,0.0019217377,0.0013808145,0.00070848106,0.0020520901,0.0023981598,0.028336342],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.96496534,0.017109986,0.0025362915,0.004299027,0.010472345,0.0006169837],"domain_scores_gemma":[0.8140928,0.10249704,0.0061681746,0.045689803,0.029546255,0.0020058483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03848623,0.0012820722,0.0016334404,0.013227209,0.003003823,0.017953645,0.0030979249,0.0027369238,0.011491465],"category_scores_gemma":[0.1011741,0.0013453459,0.0022578181,0.011088817,0.007699315,0.022847652,0.0048121223,0.005910548,0.007502023],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029149765,0.00015491588,0.0032671527,0.0021429933,0.00015724295,0.0002670238,0.011416022,0.0025191645,0.016592236,0.36380997,0.016105525,0.5832763],"study_design_scores_gemma":[0.000096111726,0.00031868217,0.0057686768,0.0027404074,0.00020082612,0.0016066916,0.0049880357,0.023219973,0.037085913,0.21484661,0.70883304,0.00029511168],"about_ca_topic_score_codex":0.004611542,"about_ca_topic_score_gemma":0.0039058826,"teacher_disagreement_score":0.03848623,"about_ca_system_score_codex":0.004630573,"about_ca_system_score_gemma":0.0049236133,"threshold_uncertainty_score":0.20353705},"labels":[],"label_agreement":null},{"id":"W2303554020","doi":"","title":"L'utilisation des POMDP pour les résumés multi-documents orientés par une thématique","year":2013,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Humanities; Political science; Philosophy; Computer science","score_opus":0.03534792806896733,"score_gpt":0.28854973517114635,"score_spread":0.253201807102179,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2303554020","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01770773,0.00095013814,0.966175,0.0013527219,0.0002925271,0.00024243673,0.0009859147,0.0050140773,0.0072793644],"genre_scores_gemma":[0.19357416,0.0010033041,0.7894008,0.00021440216,0.00010267877,0.00029491604,0.0015584447,0.0008499675,0.01300129],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99636745,0.0012734286,0.00027395226,0.0007126657,0.0011200658,0.00025249895],"domain_scores_gemma":[0.9927458,0.0050106356,0.0002269238,0.00064985413,0.0011379232,0.00022895282],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003382912,0.0012571262,0.0012530598,0.0015634097,0.0014401458,0.005494749,0.0011054919,0.0017706403,0.007230305],"category_scores_gemma":[0.014787033,0.0007692926,0.0017810117,0.0017031796,0.0011283073,0.0032707448,0.0016879877,0.00291398,0.002060437],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013118386,0.00030758022,0.0017787921,0.001752109,0.00021837991,0.0008565531,0.0018929268,0.33105835,0.022034261,0.090006925,0.012405717,0.5363766],"study_design_scores_gemma":[0.00014875559,0.00014923174,0.0005921094,0.00014517753,0.0001004538,0.00019522809,0.0005081438,0.9052029,0.021786576,0.033583876,0.037533857,0.000053685628],"about_ca_topic_score_codex":0.029597487,"about_ca_topic_score_gemma":0.026405279,"teacher_disagreement_score":0.029597487,"about_ca_system_score_codex":0.0026541082,"about_ca_system_score_gemma":0.003959349,"threshold_uncertainty_score":0.058850408},"labels":[],"label_agreement":null},{"id":"W2331950033","doi":"10.5406/amerjpsyc.127.2.0137","title":"A Remember-Know Analysis of the Semantic Serial Position Function","year":2014,"lang":"en","type":"article","venue":"The American Journal of Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Psychology; Function (biology); Serial position effect; Cognitive psychology; Semantic memory; Cognitive science; Cognition; Neuroscience; Free recall; Recall","score_opus":0.008009751395029338,"score_gpt":0.30607213291306895,"score_spread":0.2980623815180396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2331950033","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94655895,0.0001679832,0.03921862,0.0001608625,0.000027401617,0.000043375814,0.00037247987,0.00029078865,0.013159496],"genre_scores_gemma":[0.99448967,0.000041955325,0.0039527835,0.000035580306,0.00001698464,0.000022149803,0.00022877436,0.000070405215,0.0011416464],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9997502,0.00006417515,0.000023303759,0.000057555593,0.00007866572,0.000026094367],"domain_scores_gemma":[0.98622805,0.010727268,0.00062765775,0.0015236549,0.000684631,0.00020880115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015685599,0.0003912911,0.00038743528,0.0014267334,0.00026096305,0.0009217462,0.00034464203,0.00043905267,0.0064154672],"category_scores_gemma":[0.013709561,0.0002012316,0.00042982554,0.0006663863,0.0007680104,0.0027423198,0.0005021249,0.00083361263,0.0005158242],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036754587,0.0009020849,0.12458217,0.0006754803,0.00037550594,0.00091448624,0.004222668,0.009710571,0.4008485,0.100275956,0.003129209,0.35068792],"study_design_scores_gemma":[0.00017065497,0.0011370082,0.73965,0.00006196577,0.00027643298,0.002002808,0.0011194634,0.09856357,0.096069865,0.054779958,0.0060159205,0.00015242875],"about_ca_topic_score_codex":0.0005571436,"about_ca_topic_score_gemma":0.00045162652,"teacher_disagreement_score":0.0064154672,"about_ca_system_score_codex":0.00024238946,"about_ca_system_score_gemma":0.00019755744,"threshold_uncertainty_score":0.021461844},"labels":[],"label_agreement":null},{"id":"W2336430626","doi":"10.1037/apl0000108","title":"Initial investigation into computer scoring of candidate essays for personnel selection.","year":2016,"lang":"en","type":"article","venue":"Journal of Applied Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":134,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Campion College","funders":"","keywords":"PsycINFO; Dilemma; Notice; Leverage (statistics); Computer science; Data science; Psychology; Predictive validity; Context (archaeology); Scale (ratio); Selection (genetic algorithm); Personnel selection; Disadvantage; Applied psychology; Social psychology; Artificial intelligence; MEDLINE; Management; Clinical psychology","score_opus":0.021723976246193644,"score_gpt":0.3300711010410206,"score_spread":0.30834712479482695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2336430626","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.928921,0.00065446465,0.031813506,0.0035512876,0.00050306425,0.0022992971,0.0010742369,0.0010355042,0.030147668],"genre_scores_gemma":[0.9367945,0.0003705369,0.051409356,0.0006384025,0.00018687625,0.0008716279,0.0009403581,0.00012060379,0.008667771],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97932416,0.013963087,0.0012135096,0.0008930965,0.004229224,0.00037683905],"domain_scores_gemma":[0.760054,0.16959143,0.00934851,0.009653633,0.048957266,0.0023952073],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027318437,0.0006412263,0.00034478612,0.002585573,0.0013342306,0.002433471,0.001265528,0.0008376648,0.00448447],"category_scores_gemma":[0.20038082,0.00027644556,0.00031019,0.002592504,0.0007784082,0.0016209054,0.0010171831,0.0011326844,0.0018868532],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002377022,0.0029269452,0.24098933,0.00058509124,0.00009933409,0.0006964975,0.014266977,0.0024847144,0.010140191,0.004397934,0.030179562,0.6908564],"study_design_scores_gemma":[0.0006347623,0.009481611,0.71836656,0.0008540137,0.00021161033,0.0021312982,0.026494408,0.09010796,0.02653374,0.0071160514,0.11780369,0.00026431927],"about_ca_topic_score_codex":0.0035590609,"about_ca_topic_score_gemma":0.007487069,"teacher_disagreement_score":0.027318437,"about_ca_system_score_codex":0.0015043919,"about_ca_system_score_gemma":0.0020501034,"threshold_uncertainty_score":0.1444754},"labels":[],"label_agreement":null},{"id":"W2342073150","doi":"10.1108/jdoc-09-2015-0111","title":"On the composition of scientific abstracts","year":2016,"lang":"en","type":"article","venue":"Journal of Documentation","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"","keywords":"Sentence; Computer science; Rhetorical question; Argumentation theory; Relation (database); Information retrieval; Originality; Similarity (geometry); Scientific writing; Composition (language); Value (mathematics); Linguistics; Natural language processing; Artificial intelligence; Sociology; Qualitative research","score_opus":0.013616701985817723,"score_gpt":0.3029858504638016,"score_spread":0.2893691484779839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2342073150","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40476066,0.026108546,0.42926428,0.011314786,0.005802013,0.0045342664,0.012132519,0.0039005654,0.10218237],"genre_scores_gemma":[0.63578093,0.0061054174,0.32510608,0.0012151396,0.0027315859,0.0021311615,0.01250298,0.001533268,0.012893484],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9381936,0.023137463,0.010773021,0.0048560128,0.022278408,0.0007614921],"domain_scores_gemma":[0.711122,0.15962005,0.041247778,0.0143624395,0.06935618,0.0042916536],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.02917601,0.0008729821,0.0011313342,0.023388535,0.0043579503,0.009041313,0.0012152229,0.0012625067,0.0066799424],"category_scores_gemma":[0.23837289,0.0007777446,0.0010625027,0.017462224,0.0024501767,0.008899843,0.0044819615,0.0014780514,0.0032181619],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018965218,0.00022991297,0.049110565,0.00980584,0.0005174818,0.00218844,0.029371947,0.004388246,0.04107347,0.09136626,0.03846021,0.7315911],"study_design_scores_gemma":[0.00028235686,0.0011457029,0.12223194,0.004459243,0.0010557956,0.006165283,0.019013286,0.035446726,0.03834672,0.19583596,0.5755493,0.0004676744],"about_ca_topic_score_codex":0.0010953654,"about_ca_topic_score_gemma":0.000777865,"teacher_disagreement_score":0.9766115,"about_ca_system_score_codex":0.0029269643,"about_ca_system_score_gemma":0.0043636537,"threshold_uncertainty_score":0.15429932},"labels":[],"label_agreement":null},{"id":"W2342967932","doi":"10.3389/fpsyg.2016.00577","title":"Editorial: Quantum Structures in Cognitive and Social Science","year":2016,"lang":"en","type":"editorial","venue":"Frontiers in Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia","funders":"","keywords":"Psychology; Cognition; Cognitive science; Social cognition; Cognitive psychology; Neuroscience","score_opus":0.009621518641293408,"score_gpt":0.3548133138391087,"score_spread":0.3451917951978153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2342967932","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000027945573,0.0035007542,0.00013882718,0.031111805,0.9632719,0.000021328797,0.00012848162,0.00005661562,0.0017423279],"genre_scores_gemma":[0.0004730354,0.0051190513,0.00012389575,0.010043766,0.96891797,0.0000349622,0.00012304113,0.00006826406,0.015095982],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.993507,0.0008131841,0.0008324014,0.00052311685,0.003915409,0.00040895387],"domain_scores_gemma":[0.96732,0.012199072,0.0020863356,0.0010087821,0.013504938,0.0038807725],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008671004,0.0033636412,0.0049249334,0.008005107,0.0040620896,0.011148235,0.0046877437,0.012000829,0.033148948],"category_scores_gemma":[0.037240658,0.0011239359,0.0029567415,0.0045995954,0.0031607451,0.005368251,0.0028064956,0.013841352,0.020951787],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012431167,0.000004442956,0.00000893154,0.000117810414,0.000008345191,0.000039578532,0.000004928017,0.00001766104,0.000017774606,0.000213189,0.9972017,0.0023533562],"study_design_scores_gemma":[0.000057490848,0.000022731429,0.00024566968,0.0006859428,0.000045294444,0.00018122874,0.000045654826,0.00017155088,0.000113387454,0.002859266,0.9955438,0.00002791782],"about_ca_topic_score_codex":0.0029678082,"about_ca_topic_score_gemma":0.008065657,"teacher_disagreement_score":0.033148948,"about_ca_system_score_codex":0.0036145137,"about_ca_system_score_gemma":0.0060584005,"threshold_uncertainty_score":0.11089426},"labels":[],"label_agreement":null},{"id":"W2345836734","doi":"10.3758/s13423-016-1053-2","title":"The principals of meaning: Extracting semantic dimensions from co-occurrence models of semantics","year":2016,"lang":"en","type":"review","venue":"Psychonomic Bulletin & Review","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":129,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Semantics (computer science); Meaning (existential); Linguistics; Cognitive psychology; Natural language processing; Cognitive science; Epistemology; Programming language; Computer science; Psychotherapist; Philosophy","score_opus":0.07264841213761936,"score_gpt":0.37943749962019485,"score_spread":0.3067890874825755,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2345836734","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06867425,0.44621953,0.44161037,0.009688376,0.0011874676,0.00054213934,0.011488168,0.0024670432,0.018122723],"genre_scores_gemma":[0.444523,0.27716032,0.25788367,0.00071063935,0.0012830121,0.0006444208,0.014148541,0.0003889768,0.0032574125],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99916387,0.00020370384,0.00009925573,0.00015165163,0.0003439711,0.000037631577],"domain_scores_gemma":[0.9962845,0.002439925,0.0003306193,0.0002150991,0.00065909524,0.00007072277],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001933259,0.0012022983,0.0012033078,0.008381238,0.00045881688,0.002603179,0.0013586016,0.00077281764,0.0017130271],"category_scores_gemma":[0.008362569,0.00046747664,0.0015717039,0.0091491835,0.0013065023,0.0061565475,0.0012584422,0.0015570297,0.0013915992],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019492552,0.00009021748,0.009080021,0.004894876,0.0004503408,0.00012399736,0.0007769757,0.0014597097,0.0023268394,0.026084479,0.013071856,0.9414458],"study_design_scores_gemma":[0.00016868276,0.0002728623,0.060649354,0.005981073,0.0028042844,0.0020222124,0.004177792,0.062213812,0.00905074,0.56267315,0.2896394,0.00034661498],"about_ca_topic_score_codex":0.0024172533,"about_ca_topic_score_gemma":0.0035515977,"teacher_disagreement_score":0.008381238,"about_ca_system_score_codex":0.0007104997,"about_ca_system_score_gemma":0.0023902012,"threshold_uncertainty_score":0.0102241635},"labels":[],"label_agreement":null},{"id":"W2347003616","doi":"10.82308/10338","title":"SE-3D: a controlled comparative usability study of a virtual reality semantic hierarchy explorer","year":2011,"lang":"en","type":"article","venue":"eScholarship@McGill (McGill)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Information retrieval; Subject (documents); Ontology; Relevance (law); Vocabulary; Object (grammar); Visualization; World Wide Web; Usability; Ranking (information retrieval); Hierarchy; Human–computer interaction; Artificial intelligence","score_opus":0.08328705605839519,"score_gpt":0.31035432065726565,"score_spread":0.22706726459887044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2347003616","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9928203,0.00027199992,0.0021049657,0.000063102074,0.000043795357,0.0030729424,0.00023714255,0.00006784186,0.0013178281],"genre_scores_gemma":[0.96692455,0.0004842734,0.015789885,0.0003285596,0.00009209526,0.012391743,0.0006722633,0.00010091291,0.003215711],"study_design_codex":"design_other","study_design_gemma":"nonrandomized_trial","domain_scores_codex":[0.98963624,0.0061505768,0.0009968972,0.0013203493,0.0012621109,0.0006339257],"domain_scores_gemma":[0.967481,0.021587936,0.0019108162,0.0024985117,0.0053586797,0.0011630526],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0140086645,0.0015163573,0.001600155,0.0019615241,0.0012898835,0.0015807465,0.0012167738,0.0015085011,0.0032039953],"category_scores_gemma":[0.02756838,0.0007769424,0.0012986839,0.00078227796,0.0015623457,0.0020797283,0.0014550318,0.0011600029,0.00079255155],"study_design_candidate":"nonrandomized_trial","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.064398944,0.263415,0.07885593,0.013374036,0.002116156,0.002620131,0.16806506,0.002975737,0.09161328,0.0028154075,0.010957967,0.2987924],"study_design_scores_gemma":[0.019080138,0.6802092,0.19199881,0.001077349,0.0015694975,0.0012503641,0.049074925,0.005521229,0.022238944,0.0017380675,0.025482833,0.00075861317],"about_ca_topic_score_codex":0.0017779734,"about_ca_topic_score_gemma":0.0034201986,"teacher_disagreement_score":0.0140086645,"about_ca_system_score_codex":0.0009860608,"about_ca_system_score_gemma":0.0013027182,"threshold_uncertainty_score":0.07408583},"labels":[],"label_agreement":null},{"id":"W2351893935","doi":"","title":"Automatic Evaluation of Chinese Summarizations Based on Hybrid Strategy","year":2007,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"CAE (Canada)","funders":"","keywords":"Computer science; Key (lock); Point (geometry); Artificial intelligence; Foundation (evidence); Natural language processing; Base (topology); Computer security; Mathematics","score_opus":0.02315909752254468,"score_gpt":0.3441529698475553,"score_spread":0.3209938723250106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2351893935","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4793175,0.002709222,0.4946381,0.0004532835,0.00021346424,0.0014171075,0.0018192326,0.006510649,0.012921466],"genre_scores_gemma":[0.7520363,0.00048812898,0.2408887,0.00006576187,0.00008944594,0.000449776,0.0026189669,0.00020681735,0.0031561167],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.995583,0.0019331247,0.0005922214,0.00063037965,0.0010989489,0.00016225039],"domain_scores_gemma":[0.99287134,0.0022910247,0.0005046945,0.00043421125,0.0036923795,0.00020645469],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031480032,0.0011856948,0.0008973697,0.0038761382,0.0005502011,0.00186093,0.0006473927,0.00047425175,0.0023437287],"category_scores_gemma":[0.01048079,0.00018401831,0.00050037354,0.0026262165,0.00033811672,0.0021375592,0.000639529,0.00028617718,0.0006424819],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017182641,0.00025114784,0.00819078,0.0011499143,0.00040079423,0.0003615122,0.0016538077,0.011930347,0.08664891,0.008676149,0.0085644275,0.87045395],"study_design_scores_gemma":[0.00063429546,0.0031110756,0.04770218,0.00015776009,0.0013467459,0.00089243456,0.0031696393,0.72695756,0.18323407,0.010506634,0.021975748,0.00031190747],"about_ca_topic_score_codex":0.0021114533,"about_ca_topic_score_gemma":0.0021395902,"teacher_disagreement_score":0.0038761382,"about_ca_system_score_codex":0.0008186502,"about_ca_system_score_gemma":0.0007697582,"threshold_uncertainty_score":0.016648412},"labels":[],"label_agreement":null},{"id":"W2366754855","doi":"10.29173/cais375","title":"Convergence and Divergence in Tagging Systems: An Examination of Tagging Practices Over a Four Year Period","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Period (music); Divergence (linguistics); Humanities; Convergence (economics); Library science; Geography; Computer science; Art; Linguistics; Philosophy; Economics","score_opus":0.037532928269416704,"score_gpt":0.2798403217363502,"score_spread":0.2423073934669335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2366754855","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9927899,0.00062372524,0.0021378128,0.0004380156,0.000033533073,0.00006538077,0.0004460991,0.00006608632,0.0033995188],"genre_scores_gemma":[0.9939488,0.00030915454,0.002283034,0.000065503555,0.000021603726,0.00007674138,0.00083675527,0.000056126883,0.0024023368],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9882279,0.003266021,0.0015492094,0.002039866,0.00417087,0.0007460004],"domain_scores_gemma":[0.90523607,0.04104096,0.018527389,0.0066494946,0.025382124,0.003164027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01777357,0.0002911823,0.00068673043,0.009335411,0.0026422485,0.0037248838,0.001447422,0.0010255357,0.0012171923],"category_scores_gemma":[0.07208723,0.00054306706,0.00047883007,0.009958372,0.002424364,0.0046467395,0.0040923683,0.0015392686,0.00071633584],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059130834,0.00014117884,0.72924393,0.00038730822,0.00015961876,0.0007629458,0.13344842,0.00076497445,0.0063083298,0.0019876068,0.0016965705,0.12450779],"study_design_scores_gemma":[0.000007301586,0.0003291381,0.9393578,0.00010802397,0.000042277265,0.00050160673,0.042740315,0.001280033,0.0018653006,0.0005723079,0.01310545,0.000090375914],"about_ca_topic_score_codex":0.014645022,"about_ca_topic_score_gemma":0.01998457,"teacher_disagreement_score":0.01777357,"about_ca_system_score_codex":0.003215417,"about_ca_system_score_gemma":0.0015330345,"threshold_uncertainty_score":0.09399676},"labels":[],"label_agreement":null},{"id":"W2377349363","doi":"","title":"Internet Popular Topics Extraction of Traffic Content Words Correlation","year":2007,"lang":"en","type":"article","venue":"Xi'an Jiaotong Daxue xuebao","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"L'Alliance Boviteq","funders":"","keywords":"The Internet; Computer science; Cluster analysis; DBSCAN; Noise (video); Data mining; Information retrieval; World Wide Web; Artificial intelligence; Fuzzy clustering; Image (mathematics)","score_opus":0.04147398199296873,"score_gpt":0.30684495241127907,"score_spread":0.26537097041831037,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2377349363","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39259973,0.0008867136,0.59165436,0.00022691967,0.000100943485,0.00052319,0.003155202,0.00271831,0.008134633],"genre_scores_gemma":[0.8289196,0.00054413424,0.16151564,0.00003608078,0.0001197629,0.0003939598,0.0048899204,0.00012125441,0.0034596017],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987375,0.00021712665,0.000117475705,0.00029585315,0.00048643872,0.00014554843],"domain_scores_gemma":[0.99843127,0.00049253774,0.00020758428,0.00011304118,0.0006979738,0.00005755536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005700861,0.0006837008,0.0005973894,0.0063699386,0.00065748097,0.0011370032,0.00037183127,0.00046910637,0.0016586724],"category_scores_gemma":[0.0045524626,0.0002355669,0.00079768104,0.0072358525,0.0002982333,0.0015697889,0.0007051677,0.000445674,0.0012475129],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062127534,0.0002547531,0.03884385,0.0007100426,0.00019593531,0.0006849052,0.0011838463,0.010935218,0.09131586,0.010741111,0.006891272,0.837622],"study_design_scores_gemma":[0.00007622958,0.00040479418,0.12396686,0.00010086526,0.00043441495,0.0022506919,0.0019664539,0.7211074,0.11220754,0.012424885,0.024908606,0.0001512464],"about_ca_topic_score_codex":0.0027575542,"about_ca_topic_score_gemma":0.002307887,"teacher_disagreement_score":0.0063699386,"about_ca_system_score_codex":0.00047304155,"about_ca_system_score_gemma":0.0010081553,"threshold_uncertainty_score":0.005548835},"labels":[],"label_agreement":null},{"id":"W2394448961","doi":"","title":"Concept Search Engine Based on Keywords","year":2007,"lang":"en","type":"article","venue":"Microcomputer applications","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Search engine; Information retrieval; Weighting; Set (abstract data type); Order (exchange); Position (finance); Data mining; Programming language","score_opus":0.007999254089996601,"score_gpt":0.2842298550712716,"score_spread":0.276230600981275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2394448961","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034249254,0.02011371,0.7126063,0.0035435276,0.0015977933,0.003837142,0.03824021,0.045752432,0.14005955],"genre_scores_gemma":[0.15841353,0.011334454,0.7134998,0.0022836113,0.0006669644,0.002171801,0.04960917,0.0018593005,0.06016142],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981018,0.00039003833,0.0003255547,0.00032523536,0.0007282589,0.00012920459],"domain_scores_gemma":[0.99695873,0.0010978448,0.00013929274,0.00025873247,0.0013257761,0.00021967303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018352912,0.0010255123,0.0013452286,0.010246938,0.0009593346,0.0030188449,0.0013311926,0.0013377729,0.020599058],"category_scores_gemma":[0.009058884,0.00044738498,0.0010309058,0.0077407737,0.00048228345,0.007798175,0.0014819045,0.0009050097,0.012621923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009239571,0.00033775982,0.0026467827,0.00425441,0.00022713307,0.0008627624,0.0007435711,0.0033958212,0.017958947,0.12797427,0.20288198,0.6377926],"study_design_scores_gemma":[0.0005949988,0.0006600882,0.0028313752,0.0012294084,0.0004714438,0.0036320402,0.0009496487,0.09448146,0.027401078,0.14218749,0.7251806,0.0003804758],"about_ca_topic_score_codex":0.0041071298,"about_ca_topic_score_gemma":0.0034129475,"teacher_disagreement_score":0.020599058,"about_ca_system_score_codex":0.0013017463,"about_ca_system_score_gemma":0.0025547382,"threshold_uncertainty_score":0.06891072},"labels":[],"label_agreement":null},{"id":"W2395638176","doi":"10.29173/cais888","title":"Ontology-based Indexing Technologies in Information Retrieval: Building a Topic Map (ISO 13250) for a Mathematics Education Database","year":2016,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Search engine indexing; Library science; Computer science; Ontology; Information retrieval; Database; World Wide Web; Philosophy","score_opus":0.028516129917828464,"score_gpt":0.2911053196894261,"score_spread":0.2625891897715976,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2395638176","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074367723,0.0031254806,0.9610134,0.0022043078,0.00046049076,0.001146428,0.0040026587,0.007312607,0.013297766],"genre_scores_gemma":[0.041886177,0.003858066,0.93706334,0.00040043634,0.00017743034,0.0013638395,0.009761061,0.0012024541,0.0042872266],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98822093,0.0028411427,0.0025132764,0.0011961544,0.0047591305,0.0004694158],"domain_scores_gemma":[0.98889726,0.0037633246,0.0010165764,0.0023060257,0.0034972054,0.0005196848],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010771009,0.0009583941,0.0015531007,0.027213885,0.0031041764,0.010233662,0.0024728514,0.0019901271,0.0040016253],"category_scores_gemma":[0.0266554,0.0010754603,0.002533601,0.023312828,0.0019576298,0.016695073,0.0053456766,0.0020098551,0.0046397094],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019741572,0.00034460082,0.0040715924,0.0033709991,0.00025544106,0.0004684115,0.005742712,0.004156513,0.013952296,0.1413458,0.05401394,0.7720803],"study_design_scores_gemma":[0.0001530063,0.00022605878,0.008244033,0.0018359993,0.00039854,0.0014553289,0.0043236995,0.047485527,0.022060733,0.13757385,0.77593166,0.0003114816],"about_ca_topic_score_codex":0.013669894,"about_ca_topic_score_gemma":0.011798501,"teacher_disagreement_score":0.027213885,"about_ca_system_score_codex":0.00308978,"about_ca_system_score_gemma":0.006689593,"threshold_uncertainty_score":0.056963205},"labels":[],"label_agreement":null},{"id":"W2395887233","doi":"","title":"University of Waterloo at TREC 2015 Microblog Track","year":2015,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Microblogging; Exploit; Social media; Information retrieval; Relevance (law); Word (group theory); Query expansion; World Wide Web; Mathematics","score_opus":0.0386870167005127,"score_gpt":0.26424134556232476,"score_spread":0.22555432886181206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2395887233","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014678959,0.009080539,0.012938594,0.011736397,0.0029357255,0.0024574022,0.7627143,0.028921807,0.15453632],"genre_scores_gemma":[0.026019836,0.002701474,0.020794334,0.001086059,0.00039969484,0.0009350247,0.84465086,0.0016208898,0.101791725],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9957131,0.0008190097,0.00025934738,0.0009144402,0.0017987705,0.0004953261],"domain_scores_gemma":[0.9900215,0.0015181368,0.00033656452,0.001449693,0.0053407494,0.0013332699],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056370446,0.00251254,0.0022712606,0.0057795094,0.0044925753,0.005784139,0.0025498038,0.0018693382,0.0838113],"category_scores_gemma":[0.012374561,0.0012277658,0.00071722997,0.005595828,0.0011084835,0.0066098045,0.0020530606,0.0024939314,0.058410294],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000071269904,0.00008953497,0.0004119163,0.00025128457,0.000020771507,0.000027784852,0.000060829774,0.00039079197,0.00065102533,0.00077952864,0.97662896,0.020616442],"study_design_scores_gemma":[0.0003085034,0.00014359168,0.007259942,0.00028709645,0.00005167482,0.000075002936,0.00036453133,0.016177136,0.00377474,0.0034374958,0.9680072,0.00011300302],"about_ca_topic_score_codex":0.47363713,"about_ca_topic_score_gemma":0.6227235,"teacher_disagreement_score":0.47363713,"about_ca_system_score_codex":0.011152823,"about_ca_system_score_gemma":0.014058927,"threshold_uncertainty_score":0.94176054},"labels":[],"label_agreement":null},{"id":"W2398387993","doi":"","title":"CLASSY 2011 at TAC: Guided and Multi-lingual Summaries and Evaluation Metrics.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.04780621551233317,"score_gpt":0.32450437745493305,"score_spread":0.2766981619425999,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398387993","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23298977,0.0096402895,0.21361929,0.0045034536,0.0063118325,0.008292551,0.27335107,0.19532298,0.0559687],"genre_scores_gemma":[0.24256156,0.00082297367,0.26494622,0.0005975516,0.000822546,0.0052272067,0.44696116,0.012593268,0.025467612],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9653577,0.01565968,0.0042945473,0.0027172922,0.010801567,0.0011691353],"domain_scores_gemma":[0.90867954,0.026068816,0.0042179488,0.017889138,0.03859453,0.0045500267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017704917,0.0027074136,0.0020585684,0.011332618,0.0022732671,0.004768879,0.0027620082,0.0032897382,0.011935113],"category_scores_gemma":[0.08678232,0.000625378,0.0012086901,0.006980216,0.00075540453,0.0051076584,0.00371378,0.0024764661,0.010153662],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017027472,0.001084475,0.0049606697,0.0024857896,0.0006999058,0.00023353557,0.0011919227,0.007203854,0.016997267,0.0025842083,0.6663132,0.29454243],"study_design_scores_gemma":[0.0034046024,0.004986094,0.059022095,0.00091451785,0.0010286815,0.0012914597,0.0030976413,0.22838186,0.0966106,0.0125328265,0.587652,0.0010776101],"about_ca_topic_score_codex":0.012708335,"about_ca_topic_score_gemma":0.020350683,"teacher_disagreement_score":0.017704917,"about_ca_system_score_codex":0.0019811587,"about_ca_system_score_gemma":0.0036600428,"threshold_uncertainty_score":0.09363365},"labels":[],"label_agreement":null},{"id":"W2400722199","doi":"10.31234/osf.io/qs735_v1","title":"The fan effect in overlapping data sets and logical inference","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Inference; Computer science; Scope (computer science); Cognition; Cognitive architecture; Logical conjunction; Logical data model; Artificial intelligence; Psychology; Programming language; Data modeling; Database","score_opus":0.026250425279193152,"score_gpt":0.3625813832382834,"score_spread":0.33633095795909024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2400722199","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8777495,0.0013938957,0.104762785,0.00088703,0.00013255325,0.00037142806,0.00022935322,0.00056565896,0.013907814],"genre_scores_gemma":[0.9686517,0.000300865,0.029107053,0.00021781008,0.000051153405,0.00016835828,0.00014053799,0.000092556744,0.0012699445],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9911718,0.0046721376,0.0005508717,0.0010929158,0.0021105332,0.00040180562],"domain_scores_gemma":[0.56395423,0.41284755,0.008153828,0.011494141,0.0020926588,0.0014576137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017354632,0.0012459115,0.001382382,0.0012107838,0.0011784139,0.0020963775,0.0016410141,0.001592559,0.014539679],"category_scores_gemma":[0.18265973,0.0008597854,0.0011921398,0.000780466,0.004882361,0.012345108,0.0038675957,0.0031022427,0.0003893887],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.017980423,0.004644929,0.07033328,0.0055726604,0.0010418609,0.002815683,0.00964584,0.047946405,0.13051891,0.16357894,0.00427541,0.54164565],"study_design_scores_gemma":[0.0019366046,0.010105258,0.060000863,0.0007413459,0.0012168442,0.0040040165,0.002791953,0.3241455,0.14406462,0.44252262,0.007925398,0.00054497865],"about_ca_topic_score_codex":0.0017267148,"about_ca_topic_score_gemma":0.0018034711,"teacher_disagreement_score":0.017354632,"about_ca_system_score_codex":0.0009980031,"about_ca_system_score_gemma":0.0009021427,"threshold_uncertainty_score":0.0917812},"labels":[],"label_agreement":null},{"id":"W2401653906","doi":"","title":"Textual Entailment - Fitchburg State College.","year":2008,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Logical consequence; Textual entailment; Natural language processing; Artificial intelligence; Similarity (geometry); Word (group theory); Computer science; State (computer science); Mathematics; Linguistics; Algorithm; Philosophy","score_opus":0.008736516011412841,"score_gpt":0.2566495886352712,"score_spread":0.24791307262385837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2401653906","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09490702,0.005521699,0.29152262,0.014460562,0.0025767726,0.0022865785,0.061461944,0.037546672,0.48971617],"genre_scores_gemma":[0.40854445,0.0017769145,0.24258657,0.0013119457,0.001106741,0.0013699722,0.091394134,0.005149158,0.24676013],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9952024,0.0013547243,0.0004115725,0.0010059855,0.0017885874,0.00023674504],"domain_scores_gemma":[0.9889213,0.004690956,0.0004312881,0.0020314956,0.0032955993,0.00062949397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00507789,0.00094919314,0.00062655745,0.004039144,0.0019020955,0.002885587,0.001129826,0.001264077,0.15632623],"category_scores_gemma":[0.0294038,0.00054569077,0.0007926452,0.0028671273,0.00070200546,0.0041129324,0.002360304,0.0015523266,0.05005142],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009806418,0.00036669968,0.0046091354,0.00054319244,0.00006009482,0.00022820443,0.0006275326,0.0008452153,0.0051249424,0.041743983,0.24745688,0.6974135],"study_design_scores_gemma":[0.00058654364,0.00047738737,0.019095954,0.0005606751,0.00014023589,0.0012456286,0.0013289372,0.036659468,0.050153576,0.11651883,0.77308303,0.00014974587],"about_ca_topic_score_codex":0.004678562,"about_ca_topic_score_gemma":0.0064335745,"teacher_disagreement_score":0.15632623,"about_ca_system_score_codex":0.0016957853,"about_ca_system_score_gemma":0.002188819,"threshold_uncertainty_score":0.52296335},"labels":[],"label_agreement":null},{"id":"W2406374727","doi":"","title":"Is paper uncitedness a function of the alphabet","year":2015,"lang":"en","type":"article","venue":"ISSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Citation; Audience measurement; Presentation (obstetrics); Computer science; Information retrieval; Argument (complex analysis); Function (biology); Order (exchange); Visibility; Citation impact; Product (mathematics); Advertising; World Wide Web; Mathematics; Geography; Business","score_opus":0.03159327343030376,"score_gpt":0.2817840549621772,"score_spread":0.2501907815318734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2406374727","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.070325464,0.01368707,0.022097645,0.07223521,0.0601772,0.00062990625,0.03172826,0.006773409,0.7223458],"genre_scores_gemma":[0.60230863,0.012517611,0.018710403,0.016339336,0.028688012,0.0009868316,0.018418688,0.008653187,0.29337722],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9554093,0.010352765,0.0070909373,0.006308326,0.018080933,0.002757802],"domain_scores_gemma":[0.5704872,0.21892329,0.053784363,0.054975227,0.084210396,0.01761954],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.017098503,0.0007131059,0.0018436015,0.009484405,0.003406112,0.028432857,0.0026679006,0.0022457105,0.20787105],"category_scores_gemma":[0.26950178,0.0007006797,0.0007084885,0.02229665,0.0035270047,0.008887486,0.004557593,0.0021580502,0.11457037],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011104727,0.00015986971,0.02947769,0.0022178076,0.00012858369,0.000597269,0.0031516862,0.0007744952,0.0024532485,0.07158465,0.53438646,0.35395786],"study_design_scores_gemma":[0.00010658871,0.0001611933,0.023074446,0.0016041864,0.00006310254,0.00078500895,0.0030700564,0.00071440433,0.0022367823,0.04161617,0.9264892,0.000078927755],"about_ca_topic_score_codex":0.0014712429,"about_ca_topic_score_gemma":0.0007693509,"teacher_disagreement_score":0.9905156,"about_ca_system_score_codex":0.0042095734,"about_ca_system_score_gemma":0.00472378,"threshold_uncertainty_score":0.695398},"labels":[],"label_agreement":null},{"id":"W2406432149","doi":"","title":"Learning Relationship between Authors' Activity and Sentiments: A case study of online medical forums","year":2015,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Representation (politics); Focus (optics); Class (philosophy); Domain (mathematical analysis); Conditional random field; Support vector machine; Data science; Natural language processing; Artificial intelligence; Information retrieval; World Wide Web; Mathematics","score_opus":0.08065363470453168,"score_gpt":0.390861903818522,"score_spread":0.3102082691139903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2406432149","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9959908,0.00017327846,0.0025123686,0.00022225664,0.000015553533,0.000028872997,0.00026926672,0.00003605285,0.00075157196],"genre_scores_gemma":[0.9931356,0.00009907099,0.005715056,0.00004144816,0.000045389894,0.000024439421,0.00045184468,0.00001856259,0.0004687049],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9971782,0.0018138498,0.00016163693,0.0003114228,0.0004107985,0.000124062],"domain_scores_gemma":[0.9423205,0.04951068,0.0036854672,0.0010602347,0.0023232978,0.001099816],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037154763,0.00038580497,0.00032208287,0.0023257001,0.0008621016,0.001095125,0.00044708845,0.0010266257,0.0008674452],"category_scores_gemma":[0.024221957,0.00018198567,0.00038222247,0.001840673,0.0006011265,0.0018165844,0.0006706972,0.00061989715,0.00041153765],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083749794,0.001458576,0.83022684,0.00048566295,0.00016496454,0.0032787567,0.017588226,0.0028859912,0.015139847,0.0008414877,0.003927507,0.12316461],"study_design_scores_gemma":[0.00019426945,0.001358767,0.79540527,0.00023801663,0.00025852775,0.00491036,0.03127392,0.1250312,0.018371105,0.004962067,0.017875355,0.00012124247],"about_ca_topic_score_codex":0.0013237868,"about_ca_topic_score_gemma":0.0036219745,"teacher_disagreement_score":0.0037154763,"about_ca_system_score_codex":0.00040928702,"about_ca_system_score_gemma":0.0003265572,"threshold_uncertainty_score":0.019649506},"labels":[],"label_agreement":null},{"id":"W2407694321","doi":"","title":"SemQuest: University of Houston's Semantics-based Question Answering System.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Question answering; Computer science; Preprocessor; Redundancy (engineering); Natural language processing; Semantics (computer science); Relevance (law); Extractor; Sentence; Artificial intelligence; Information retrieval; Programming language; Engineering","score_opus":0.007996082422250674,"score_gpt":0.22139675349799368,"score_spread":0.213400671075743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407694321","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027529856,0.002522188,0.27742782,0.0027472286,0.0005500607,0.0014360046,0.14513575,0.48716098,0.055490095],"genre_scores_gemma":[0.15219428,0.0013527512,0.49314213,0.0015166661,0.00034520513,0.0015474752,0.29709244,0.009884541,0.042924616],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99931014,0.00021649024,0.000068448586,0.0001555115,0.00020025912,0.000049136816],"domain_scores_gemma":[0.99846953,0.000547258,0.0001519464,0.00020412449,0.00047861258,0.00014849537],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015797616,0.0011419115,0.00062193023,0.0025306852,0.00064330414,0.0017015468,0.0012080283,0.0009751446,0.027969817],"category_scores_gemma":[0.005381971,0.00048406332,0.0005204605,0.0012898828,0.00031366668,0.0037784723,0.0016659835,0.0008609516,0.018424114],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066262065,0.0001814772,0.0028051704,0.001028364,0.000072778625,0.0004018848,0.0008390805,0.001913911,0.012341669,0.0075907116,0.68866545,0.28349686],"study_design_scores_gemma":[0.0004055326,0.0002898643,0.006606184,0.00020112006,0.000075098484,0.0006781439,0.000839828,0.048635576,0.026048437,0.031082608,0.88501495,0.00012258255],"about_ca_topic_score_codex":0.0026556626,"about_ca_topic_score_gemma":0.0045582377,"teacher_disagreement_score":0.027969817,"about_ca_system_score_codex":0.0007410785,"about_ca_system_score_gemma":0.0013287726,"threshold_uncertainty_score":0.093568325},"labels":[],"label_agreement":null},{"id":"W2413001881","doi":"10.29173/cais250","title":"A Method for Comparing Large Scale Inter-indexer Consistency Using IR Modeling","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Mathematics; Consistency (knowledge bases); Humanities; Geometry; Philosophy","score_opus":0.05886262153393068,"score_gpt":0.3156789201341757,"score_spread":0.256816298600245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2413001881","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0046298206,0.00013411959,0.99235415,0.000064747226,0.000045115663,0.00015202624,0.00029131988,0.0014880582,0.00084054697],"genre_scores_gemma":[0.09994636,0.00010299052,0.8958885,0.00008306246,0.000067426336,0.00069674145,0.0010290986,0.0007277789,0.0014580378],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9780782,0.00890335,0.0015413808,0.004262894,0.006637405,0.0005767216],"domain_scores_gemma":[0.9460822,0.032225482,0.00444471,0.009836964,0.0069415853,0.00046906935],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02151863,0.0016909046,0.0020087815,0.008144176,0.0015142902,0.005483837,0.0033730376,0.0021734547,0.005037557],"category_scores_gemma":[0.07702534,0.0010933372,0.0024800769,0.008839519,0.0017063901,0.0051525203,0.0030979365,0.003182712,0.0026642631],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010396396,0.00041101352,0.024327269,0.0007145441,0.0016837913,0.0001853304,0.0017756774,0.08016898,0.020874184,0.047574505,0.00974342,0.8115017],"study_design_scores_gemma":[0.00019449889,0.0007365956,0.015799476,0.00013610128,0.00046765083,0.00063942367,0.00073419674,0.84577537,0.022287019,0.091306716,0.02154503,0.00037804618],"about_ca_topic_score_codex":0.0039857174,"about_ca_topic_score_gemma":0.0033677183,"teacher_disagreement_score":0.02151863,"about_ca_system_score_codex":0.0014360606,"about_ca_system_score_gemma":0.002761911,"threshold_uncertainty_score":0.11380279},"labels":[],"label_agreement":null},{"id":"W2415886958","doi":"10.29173/cais353","title":"Comparison of the Effectiveness of Related Functions in Web of Science and Scopus","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Scopus; Web of science; Function (biology); Humanities; Philosophy; Political science; MEDLINE; Biology","score_opus":0.023271217397422186,"score_gpt":0.2865563339678965,"score_spread":0.2632851165704743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2415886958","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9452977,0.00821259,0.012373004,0.0008276003,0.00008612976,0.00029260071,0.0022080669,0.001962581,0.028739681],"genre_scores_gemma":[0.9841181,0.0019415763,0.009235156,0.000093399314,0.00006985799,0.000101848964,0.002080074,0.00021868244,0.002141361],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9856872,0.006230043,0.0017376422,0.0008986032,0.0048236963,0.0006227589],"domain_scores_gemma":[0.8393837,0.13539928,0.007917379,0.004733575,0.009838009,0.0027280543],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.012220012,0.0010527209,0.0014509265,0.027253803,0.0007609458,0.0044562696,0.00078532303,0.0010472813,0.003193046],"category_scores_gemma":[0.09744451,0.0002335619,0.0016438954,0.015224446,0.00086483435,0.0064047524,0.0016479675,0.0005847691,0.0013664211],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032842052,0.0009429771,0.29369393,0.0047856877,0.0017057165,0.0004779888,0.0020470014,0.009415357,0.0067241606,0.003327409,0.0062289573,0.6673666],"study_design_scores_gemma":[0.00033253475,0.0042930082,0.8576447,0.001108815,0.0023901162,0.0016510732,0.006438284,0.07074397,0.027696645,0.006284462,0.021114992,0.0003014274],"about_ca_topic_score_codex":0.0028193884,"about_ca_topic_score_gemma":0.0024293037,"teacher_disagreement_score":0.98778,"about_ca_system_score_codex":0.0012017196,"about_ca_system_score_gemma":0.0012166365,"threshold_uncertainty_score":0.064626336},"labels":[],"label_agreement":null},{"id":"W2460693621","doi":"10.29173/cais401","title":"Inedxing as Problem Solving: A Cognitive Approach to Consistency","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Consistency (knowledge bases); Cognition; Psychology; Humanities; Cognitive psychology; Computer science; Philosophy; Artificial intelligence","score_opus":0.020619320318016783,"score_gpt":0.25436363081794305,"score_spread":0.23374431049992628,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2460693621","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5132497,0.0012697036,0.4278932,0.0028572418,0.00009098048,0.000570975,0.00012652969,0.00038034914,0.0535614],"genre_scores_gemma":[0.8912634,0.00036481133,0.10589691,0.00032153522,0.00006537501,0.0004209054,0.000093374016,0.000051838007,0.0015218742],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9872388,0.006724567,0.00074291875,0.0013932717,0.0035605065,0.00033981187],"domain_scores_gemma":[0.9380382,0.04693516,0.007200436,0.003945864,0.0031203069,0.00076011155],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010634545,0.0007077334,0.0005259604,0.005000251,0.0012153665,0.008569456,0.002358314,0.0015219583,0.002302863],"category_scores_gemma":[0.075673945,0.00044356656,0.0007888926,0.003245826,0.013218237,0.00815341,0.0040343953,0.0022194996,0.00018110026],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005090502,0.00077389204,0.07913498,0.0012186404,0.00035660758,0.0006206498,0.16075236,0.007985887,0.018932017,0.5043077,0.0010457706,0.22436237],"study_design_scores_gemma":[0.00021376356,0.00076450413,0.09140287,0.00044858005,0.0002972039,0.0013335337,0.04749895,0.0674883,0.012655449,0.7610587,0.016537638,0.00030041544],"about_ca_topic_score_codex":0.002669736,"about_ca_topic_score_gemma":0.0018499161,"teacher_disagreement_score":0.010634545,"about_ca_system_score_codex":0.0021380098,"about_ca_system_score_gemma":0.0018521696,"threshold_uncertainty_score":0.056241512},"labels":[],"label_agreement":null},{"id":"W2463112553","doi":"10.1145/2911451.2914704","title":"Simple Dynamic Emission Strategies for Microblog Filtering","year":2016,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Microblogging; Computer science; Social media; Simple (philosophy); Key (lock); Human–computer interaction; World Wide Web; Computer security","score_opus":0.01252131438409449,"score_gpt":0.3000183106035962,"score_spread":0.2874969962195017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2463112553","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055204157,0.0011863159,0.9320048,0.00037807,0.00013842295,0.00028643553,0.00021178763,0.005954312,0.0046357103],"genre_scores_gemma":[0.69135314,0.00049256854,0.29987642,0.00029188438,0.00017717382,0.000311012,0.00036130205,0.00061684946,0.006519592],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99686843,0.0009922939,0.00024815399,0.00057176227,0.0010190228,0.00030041955],"domain_scores_gemma":[0.9886204,0.007793109,0.00050037063,0.0017106956,0.0009924441,0.0003829504],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035934777,0.0014715357,0.0015483444,0.0018143894,0.0014516832,0.0021656137,0.002478828,0.0018346198,0.0034022098],"category_scores_gemma":[0.017563928,0.0005793988,0.00072941434,0.0012876512,0.0011582468,0.0050206804,0.0020975869,0.001795667,0.0015902551],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011329694,0.0012161933,0.005700178,0.0006650923,0.00020968083,0.00040796623,0.0020594837,0.08920872,0.084163375,0.052385155,0.01111068,0.75174046],"study_design_scores_gemma":[0.00015695149,0.0003033443,0.0020663165,0.000064859545,0.00011905321,0.00045956724,0.00037697356,0.8865512,0.03637603,0.0585639,0.014814867,0.00014696314],"about_ca_topic_score_codex":0.0025937755,"about_ca_topic_score_gemma":0.0033219352,"teacher_disagreement_score":0.0035934777,"about_ca_system_score_codex":0.00097214227,"about_ca_system_score_gemma":0.0012000244,"threshold_uncertainty_score":0.019004345},"labels":[],"label_agreement":null},{"id":"W2470107876","doi":"10.29173/cais728","title":"Using Genetics in Information Filtering","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Parsing; Cognition; Computer science; Cognitive science; Information filtering system; Artificial intelligence; Natural language processing; Information retrieval; Psychology","score_opus":0.02763279480462613,"score_gpt":0.2640034070986742,"score_spread":0.23637061229404807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2470107876","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017420825,0.0014683473,0.9708189,0.002357858,0.0001316088,0.000051644536,0.000052860476,0.00031406802,0.0073839445],"genre_scores_gemma":[0.42389143,0.0023500456,0.56303155,0.001400951,0.00033046285,0.00023937841,0.00019272052,0.00021334983,0.008350109],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99837434,0.0008359587,0.00007254873,0.00038221586,0.0002705402,0.000064329746],"domain_scores_gemma":[0.9942932,0.004699403,0.00023029986,0.00034075245,0.00033078334,0.00010555808],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030331034,0.00064886286,0.0008085488,0.0015778419,0.0009306308,0.002319239,0.0009242861,0.001433238,0.0031297032],"category_scores_gemma":[0.011172672,0.00039495778,0.0010198812,0.001083837,0.0038951945,0.0025936153,0.00123204,0.0012421121,0.00046466483],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009941013,0.000089565976,0.0070450907,0.00025770743,0.00023519642,0.00037178094,0.0011313964,0.21051499,0.0042547225,0.5553408,0.00308706,0.21757218],"study_design_scores_gemma":[0.000045154116,0.000077734796,0.0013417036,0.00007440028,0.00006434977,0.00017573308,0.000103401006,0.32520458,0.0017708674,0.6552942,0.015798736,0.000049130955],"about_ca_topic_score_codex":0.009205051,"about_ca_topic_score_gemma":0.008042491,"teacher_disagreement_score":0.009205051,"about_ca_system_score_codex":0.0022365523,"about_ca_system_score_gemma":0.0012582612,"threshold_uncertainty_score":0.018302917},"labels":[],"label_agreement":null},{"id":"W2473169082","doi":"","title":"Fuzziness for classification and visual query interface: platform independent query model with self-adaptive fuzzy capabilities","year":2011,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Fuzzy logic; Data mining; Classifier (UML); Machine learning; Artificial intelligence; The Internet; Fuzzy set; Interface (matter); Information retrieval; World Wide Web","score_opus":0.04484659630979332,"score_gpt":0.2836074568777472,"score_spread":0.23876086056795387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2473169082","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01767096,0.00029734694,0.9731662,0.00062102795,0.000049191814,0.00026367005,0.00016521056,0.0012087356,0.006557664],"genre_scores_gemma":[0.5179042,0.0005768719,0.46784785,0.00039696592,0.00008226502,0.00071496505,0.0005727728,0.0001770395,0.0117270835],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99785656,0.00041873165,0.00017160048,0.0004927545,0.00089586724,0.00016443022],"domain_scores_gemma":[0.9983485,0.0006512079,0.00013324169,0.00033372157,0.00045194488,0.00008144319],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022740178,0.00065379025,0.0006178601,0.0010293704,0.0007086016,0.004301268,0.0025433092,0.0019115955,0.0030057344],"category_scores_gemma":[0.0053139185,0.0004325669,0.0014531175,0.00091280945,0.0017144129,0.0047807186,0.0014424648,0.0020432023,0.00094550324],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006483571,0.00023622692,0.002622863,0.0003407751,0.00009907676,0.00075618207,0.0034418309,0.1384549,0.035476238,0.6745929,0.0073427823,0.13598786],"study_design_scores_gemma":[0.000035886897,0.00010893273,0.0004253485,0.00005387791,0.00004481561,0.000219051,0.00015334078,0.9117292,0.006035007,0.06991122,0.0112419035,0.00004136667],"about_ca_topic_score_codex":0.008187495,"about_ca_topic_score_gemma":0.0031765623,"teacher_disagreement_score":0.008187495,"about_ca_system_score_codex":0.0020505951,"about_ca_system_score_gemma":0.0014302528,"threshold_uncertainty_score":0.016279697},"labels":[],"label_agreement":null},{"id":"W2495878820","doi":"10.29173/cais858","title":"Information Behavior Research: Where Have We Been, Where Are We Going?","year":2016,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Sociology; Political science; Philosophy","score_opus":0.0691220101564011,"score_gpt":0.3229137776906782,"score_spread":0.2537917675342771,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2495878820","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46824306,0.1653169,0.042946495,0.21792546,0.003201218,0.00086348155,0.005336636,0.00060842803,0.09555837],"genre_scores_gemma":[0.8846074,0.060650114,0.030159038,0.01008044,0.0017893423,0.00103459,0.0019627786,0.00037012072,0.009346175],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9764631,0.0125817405,0.002164708,0.0015429861,0.0060070823,0.0012402711],"domain_scores_gemma":[0.81638396,0.11289747,0.016667338,0.007455722,0.04019117,0.00640444],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.039291795,0.0004255364,0.0009421993,0.015283948,0.003612917,0.017739315,0.0013000043,0.0023833718,0.0040078065],"category_scores_gemma":[0.09940849,0.000517196,0.00077366247,0.021346692,0.006134146,0.02712688,0.0021203172,0.0026177743,0.0015013666],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044670753,0.00030508862,0.24386214,0.0076475404,0.00027685484,0.0002332191,0.09963448,0.00054720766,0.004412789,0.07968322,0.025850782,0.53709996],"study_design_scores_gemma":[0.00003285805,0.0005791257,0.4742633,0.013074276,0.00018544724,0.0007174952,0.2122116,0.0014786491,0.004026782,0.03584737,0.2573059,0.00027715813],"about_ca_topic_score_codex":0.010479247,"about_ca_topic_score_gemma":0.009197469,"teacher_disagreement_score":0.9607082,"about_ca_system_score_codex":0.009964655,"about_ca_system_score_gemma":0.010549946,"threshold_uncertainty_score":0.20779735},"labels":[],"label_agreement":null},{"id":"W2500612743","doi":"10.4018/978-1-60960-040-2.ch011","title":"Exploring Virtual Communities with the Internet Community Text Analyzer (ICTA)","year":2011,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; The Internet; World Wide Web; Set (abstract data type); Visualization; Focus (optics); Virtual community; Social network (sociolinguistics); Interface (matter); Data science; Social network analysis; Social media; Data mining","score_opus":0.09994772145408938,"score_gpt":0.2603172478232629,"score_spread":0.1603695263691735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2500612743","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05315955,0.0022458148,0.8738318,0.0009819536,0.00019441884,0.00047304644,0.0055685444,0.027628498,0.03591636],"genre_scores_gemma":[0.08580785,0.0009581718,0.89768165,0.000106543965,0.00007139469,0.0003811226,0.003702518,0.0014230588,0.009867649],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966335,0.00008449762,0.000019520678,0.000078178986,0.00013775486,0.000016671433],"domain_scores_gemma":[0.99921477,0.0005341732,0.00006188664,0.00006182328,0.000084001986,0.000043341337],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005747196,0.00077840785,0.00040114875,0.0051594665,0.0008276193,0.0025471759,0.0006866199,0.0005235141,0.011675202],"category_scores_gemma":[0.001939605,0.00035677024,0.0005136778,0.0034442178,0.00034774537,0.003413796,0.0013075854,0.0007286428,0.004377876],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011266135,0.00014794995,0.0043981075,0.000653062,0.0000928321,0.00046776352,0.0029703567,0.004744254,0.02537173,0.037887555,0.048685562,0.8744681],"study_design_scores_gemma":[0.00009566955,0.00015255256,0.0154292,0.0004684707,0.00012723761,0.0027517958,0.003917369,0.42472675,0.03362264,0.15378469,0.3647352,0.00018848958],"about_ca_topic_score_codex":0.0009493916,"about_ca_topic_score_gemma":0.0018033467,"teacher_disagreement_score":0.011675202,"about_ca_system_score_codex":0.0003994948,"about_ca_system_score_gemma":0.00048214215,"threshold_uncertainty_score":0.039057493},"labels":[],"label_agreement":null},{"id":"W2507709051","doi":"10.1145/2960811.2960813","title":"Relaxing Orthogonality Assumption in Conceptual Text Document Similarity","year":2016,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Orthogonality; Similarity (geometry); Cosine similarity; Information retrieval; Computer science; Space (punctuation); Measure (data warehouse); Similarity measure; Key (lock); Natural language processing; Artificial intelligence; Data mining; Mathematics; Pattern recognition (psychology)","score_opus":0.01983061498926167,"score_gpt":0.2893450317022357,"score_spread":0.26951441671297405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2507709051","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05363547,0.0011154066,0.9385213,0.00073319103,0.00022558334,0.00013797643,0.000449594,0.00018054826,0.005000849],"genre_scores_gemma":[0.72023004,0.0023688185,0.26942316,0.0005731027,0.0009327341,0.00069570506,0.0018896208,0.00014912427,0.0037376457],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98300123,0.0056971987,0.0016899256,0.004465098,0.0044432413,0.00070332736],"domain_scores_gemma":[0.96895164,0.015827898,0.0026116134,0.007168867,0.004862701,0.0005773163],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008423096,0.00075700466,0.0012541226,0.0034001218,0.0013634702,0.003741766,0.001717226,0.0012808882,0.0023098742],"category_scores_gemma":[0.056524478,0.00056095957,0.0014104244,0.0061189113,0.0033702417,0.009865261,0.0049877353,0.003496836,0.001138115],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005314193,0.00023281542,0.012101262,0.00072460953,0.00035451833,0.0005710213,0.0021903757,0.024956465,0.013454434,0.67634475,0.0050173164,0.26352096],"study_design_scores_gemma":[0.000103025006,0.00038262922,0.0069152517,0.00018213608,0.00020886645,0.0013244181,0.0009880337,0.21817616,0.0071924524,0.7450852,0.019290553,0.00015123952],"about_ca_topic_score_codex":0.0020534485,"about_ca_topic_score_gemma":0.0013264953,"teacher_disagreement_score":0.008423096,"about_ca_system_score_codex":0.0009732803,"about_ca_system_score_gemma":0.0019376194,"threshold_uncertainty_score":0.044546068},"labels":[],"label_agreement":null},{"id":"W251035542","doi":"10.1007/0-306-47542-1_12","title":"Learning Negotiations with Web-Based Systems","year":2006,"lang":"en","type":"book-chapter","venue":"Kluwer Academic Publishers eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; International Institute for Applied Systems Analysis","keywords":"Negotiation; World Wide Web; Computer science; Political science; Law","score_opus":0.011691515903564164,"score_gpt":0.226640238797568,"score_spread":0.21494872289400385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W251035542","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025556512,0.0029473838,0.23169707,0.006710348,0.00027878815,0.000110735426,0.000119403994,0.0007426743,0.73183703],"genre_scores_gemma":[0.634512,0.0034794388,0.069940306,0.00057839946,0.0003863929,0.00032878964,0.00040384277,0.0003910727,0.28997967],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99753964,0.0013664218,0.000100946076,0.00017640265,0.0006686566,0.00014797937],"domain_scores_gemma":[0.99501204,0.0039376025,0.00016260489,0.000458898,0.00020576641,0.00022300825],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002463411,0.00045991575,0.00040051425,0.00062812364,0.0017832316,0.008427079,0.0014457274,0.0023807485,0.055418186],"category_scores_gemma":[0.009737681,0.00057118415,0.0003435405,0.0015864243,0.0021942235,0.014512552,0.0034426406,0.0021545866,0.0089561995],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010362247,0.00021264321,0.00064712676,0.00020898921,0.000022224658,0.00022798996,0.004395156,0.011569512,0.0011256047,0.737784,0.022026407,0.2216767],"study_design_scores_gemma":[0.000040496696,0.000040345047,0.0003118321,0.00017390233,0.000011979834,0.00013033094,0.0022540207,0.028773725,0.0015996414,0.7920179,0.17462474,0.000021260203],"about_ca_topic_score_codex":0.0007657511,"about_ca_topic_score_gemma":0.0009021771,"teacher_disagreement_score":0.055418186,"about_ca_system_score_codex":0.0010785861,"about_ca_system_score_gemma":0.0011431805,"threshold_uncertainty_score":0.18539232},"labels":[],"label_agreement":null},{"id":"W2512574971","doi":"10.1002/9781119159704.ch2","title":"The Nature‐Science Approach: Some Further Consequences","year":2016,"lang":"en","type":"other","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University; Emera (Canada)","funders":"","keywords":"Natural (archaeology); Cognitive dissonance; Natural science; Energy (signal processing); Characterization (materials science); Set (abstract data type); Natural organic matter; Epistemology; Computer science; Psychology; Environmental science; Mathematics; Physics; Geography; Philosophy; Social psychology","score_opus":0.010198594719120541,"score_gpt":0.28774775309113343,"score_spread":0.2775491583720129,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2512574971","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011604748,0.016869215,0.10783875,0.12222188,0.002679569,0.000082505125,0.00019815256,0.00012560366,0.7383796],"genre_scores_gemma":[0.7274234,0.02017225,0.077358946,0.015535941,0.007891599,0.00051708886,0.0003126922,0.00041145712,0.15037659],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9976482,0.0012149743,0.00008380944,0.00024432765,0.0006918049,0.00011690056],"domain_scores_gemma":[0.9954503,0.0031402041,0.000094903386,0.00044724334,0.000703783,0.00016353562],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036257193,0.00083212624,0.00078898884,0.0018286392,0.0029782278,0.006773112,0.0014815447,0.0028791598,0.020631555],"category_scores_gemma":[0.0052029,0.0003281194,0.0008031712,0.0020825872,0.018837113,0.01585777,0.0032352235,0.0061273817,0.0012589656],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000033981576,0.000008884339,0.000022822105,0.000009589702,0.0000012921652,0.000016817265,0.00014628499,0.00009662009,0.00003567679,0.9953668,0.001831326,0.0024604416],"study_design_scores_gemma":[0.000003397757,0.0000022330646,0.000051498897,0.000010554675,7.6723825e-7,0.000016499846,0.00016632357,0.00034849704,0.000041838968,0.98208916,0.017265845,0.0000034401571],"about_ca_topic_score_codex":0.0043218634,"about_ca_topic_score_gemma":0.003956394,"teacher_disagreement_score":0.020631555,"about_ca_system_score_codex":0.0044962536,"about_ca_system_score_gemma":0.0020834624,"threshold_uncertainty_score":0.06901944},"labels":[],"label_agreement":null},{"id":"W2521029362","doi":"10.1045/september2016-atanassova","title":"Temporal Properties of Recurring In-text References","year":2016,"lang":"en","type":"article","venue":"D-Lib Magazine","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Section (typography); Rhetorical question; Computer science; Linguistics; History; Philosophy","score_opus":0.028875440863421056,"score_gpt":0.25641390908519507,"score_spread":0.227538468221774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2521029362","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5981218,0.0025463698,0.3475647,0.0009305308,0.00034901674,0.00022958752,0.008178716,0.003479443,0.03859983],"genre_scores_gemma":[0.95649624,0.00042378067,0.029243879,0.000099047626,0.00025361803,0.00011690178,0.004419144,0.0006384566,0.008308874],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.996172,0.00056560367,0.00044803778,0.0006520425,0.0018434199,0.00031890813],"domain_scores_gemma":[0.9448398,0.03288911,0.006142838,0.004707161,0.010273132,0.0011480321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030898857,0.00033345676,0.00050916744,0.006529007,0.0012849413,0.0028404696,0.0013770321,0.0006981795,0.008318846],"category_scores_gemma":[0.041531403,0.0004253989,0.0004540382,0.0058653965,0.0007592414,0.004602716,0.0010318343,0.0008455612,0.0015571363],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023061612,0.00033947613,0.061032835,0.0013955077,0.0002769792,0.0036509517,0.0064675766,0.015634542,0.06995424,0.5073476,0.01673873,0.31485543],"study_design_scores_gemma":[0.00015593022,0.00050510024,0.067686975,0.0005418676,0.0006373694,0.004514474,0.0034740816,0.37528893,0.06442834,0.40545118,0.077112414,0.0002033638],"about_ca_topic_score_codex":0.002901404,"about_ca_topic_score_gemma":0.0031474405,"teacher_disagreement_score":0.008318846,"about_ca_system_score_codex":0.0010389725,"about_ca_system_score_gemma":0.0009382148,"threshold_uncertainty_score":0.02782929},"labels":[],"label_agreement":null},{"id":"W2523945865","doi":"10.18438/b8r90p","title":"Determining Gate Count Reliability in a Library Setting","year":2016,"lang":"en","type":"article","venue":"Evidence Based Library and Information Practice","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Turnstile; Reliability (semiconductor); Computer science; Data collection; Set (abstract data type); Statistics; Mathematics; Physics","score_opus":0.008833506208491802,"score_gpt":0.25854027912697075,"score_spread":0.24970677291847895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2523945865","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9653225,0.00035510075,0.018446311,0.0002825187,0.0001031128,0.00027267679,0.0014347525,0.0010961953,0.012686848],"genre_scores_gemma":[0.98578346,0.00014967658,0.010362081,0.00015605219,0.0000649813,0.00019322505,0.0010539066,0.00016063912,0.002076033],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9689172,0.012941991,0.0021134745,0.0033526816,0.011083509,0.0015910632],"domain_scores_gemma":[0.93412304,0.01980399,0.010192414,0.0066928705,0.027591635,0.0015961082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016291028,0.00060860644,0.0007512878,0.0033665532,0.0015638652,0.0031863055,0.0025653178,0.0007571233,0.004446296],"category_scores_gemma":[0.06582866,0.00063537015,0.0006572652,0.004401266,0.0010848995,0.0016137648,0.0034006003,0.00091270206,0.0040530595],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015168741,0.0004063035,0.85426813,0.0003174799,0.0002791687,0.0006549316,0.005859836,0.0043338505,0.0049184933,0.0009878003,0.0088682575,0.11758894],"study_design_scores_gemma":[0.0000845037,0.0033623688,0.9204689,0.00020660459,0.00034539856,0.0014423663,0.007736166,0.02212756,0.022803595,0.0013799561,0.019795822,0.00024681038],"about_ca_topic_score_codex":0.014002218,"about_ca_topic_score_gemma":0.015863687,"teacher_disagreement_score":0.016291028,"about_ca_system_score_codex":0.0032101825,"about_ca_system_score_gemma":0.0019777298,"threshold_uncertainty_score":0.08615619},"labels":[],"label_agreement":null},{"id":"W2526489827","doi":"10.7202/1028615ar","title":"L’analyse de références bibliographiques assistée par ordinateur","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Hydro-Québec","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.055436402563918165,"score_gpt":0.37142323677003247,"score_spread":0.3159868342061143,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2526489827","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23889267,0.01430505,0.57543427,0.00897924,0.0013750453,0.0014869032,0.029491253,0.021987123,0.10804834],"genre_scores_gemma":[0.33501244,0.0069314893,0.5789417,0.0007952294,0.00048918475,0.0011236918,0.02429193,0.003786264,0.04862811],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.97974247,0.0054838955,0.002291925,0.002095131,0.009810897,0.00057563453],"domain_scores_gemma":[0.92359895,0.03658896,0.004044879,0.00915378,0.025986072,0.00062726595],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.011993103,0.00093811285,0.0011775359,0.020172993,0.0018743942,0.008317843,0.0013379176,0.0012567437,0.01182169],"category_scores_gemma":[0.07123157,0.00060533325,0.0010987389,0.024903186,0.0011883542,0.006596799,0.0027849888,0.0015145283,0.008024408],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040252693,0.0001120941,0.026283449,0.0043453467,0.0003597664,0.0013650453,0.025600553,0.0026327989,0.05231859,0.026439726,0.030619338,0.8295207],"study_design_scores_gemma":[0.000068479654,0.0002155076,0.05126356,0.0022687835,0.00058194826,0.0020449064,0.017809052,0.014992742,0.0909128,0.019072844,0.80041295,0.0003564078],"about_ca_topic_score_codex":0.015936261,"about_ca_topic_score_gemma":0.017645216,"teacher_disagreement_score":0.9916822,"about_ca_system_score_codex":0.0022848262,"about_ca_system_score_gemma":0.0057996446,"threshold_uncertainty_score":0.063426316},"labels":[],"label_agreement":null},{"id":"W2530746622","doi":"10.1016/b978-0-12-804412-4.00011-5","title":"Opinion Summarization and Visualization","year":2016,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of the Fraser Valley","funders":"","keywords":"Automatic summarization; Visualization; Computer science; Variety (cybernetics); Social media; Data science; Data visualization; Information retrieval; Multi-document summarization; World Wide Web; Information visualization; Artificial intelligence","score_opus":0.013690071245827141,"score_gpt":0.2738064960185678,"score_spread":0.26011642477274066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2530746622","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062450166,0.009240078,0.80126727,0.0024938982,0.0017446126,0.00027199375,0.004020043,0.020169059,0.15454794],"genre_scores_gemma":[0.08719611,0.011538038,0.6024335,0.0007834163,0.0022573443,0.00054297945,0.011195187,0.00537513,0.27867824],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995838,0.0001026634,0.000034575074,0.00008846374,0.00016274789,0.000027799713],"domain_scores_gemma":[0.99931145,0.00027053495,0.00004522831,0.00011343524,0.0002222272,0.00003709973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006163659,0.0012062662,0.0006506742,0.0021660917,0.00046660606,0.0034463038,0.00073991873,0.0006888444,0.07592613],"category_scores_gemma":[0.0025734173,0.00037421955,0.00070460566,0.0031043093,0.00036591635,0.002580332,0.0012988255,0.00089714624,0.02753875],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049292936,0.000019612837,0.00011369777,0.0004294653,0.000025027755,0.000060869752,0.00030025904,0.0012521668,0.008828858,0.014113038,0.14822136,0.8265863],"study_design_scores_gemma":[0.000045800552,0.00009091856,0.0018602025,0.0004373081,0.000087286724,0.0004220365,0.0006153169,0.057480857,0.020064956,0.11197111,0.8068471,0.00007705642],"about_ca_topic_score_codex":0.00079409947,"about_ca_topic_score_gemma":0.0010025678,"teacher_disagreement_score":0.07592613,"about_ca_system_score_codex":0.00036905744,"about_ca_system_score_gemma":0.0003787874,"threshold_uncertainty_score":0.25399822},"labels":[],"label_agreement":null},{"id":"W2534966764","doi":"10.22374/cjgim.v11i2.142","title":"Inadequate Presentation of Evidence in an Internal Medicine Conference","year":2016,"lang":"en","type":"article","venue":"Canadian Journal of General Internal Medicine","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Medicine; Presentation (obstetrics); Number needed to treat; Absolute (philosophy); Relative risk; Frequency; Family medicine; Internal medicine; Statistics; Epistemology; Mathematics; Surgery; Philosophy","score_opus":0.12257900290233292,"score_gpt":0.37820517653575025,"score_spread":0.25562617363341733,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2534966764","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4580268,0.03954677,0.042503465,0.34128162,0.016659614,0.0037116108,0.001853865,0.0018940319,0.094522156],"genre_scores_gemma":[0.9057823,0.01624575,0.034577284,0.029328924,0.0050780997,0.0018178975,0.0007583514,0.0002886605,0.0061227735],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.8870163,0.07478067,0.0120941065,0.0028492757,0.02068053,0.0025791666],"domain_scores_gemma":[0.5050726,0.3796985,0.044697683,0.008449259,0.049222697,0.012859313],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09813248,0.00071884826,0.0010427679,0.0035581212,0.0030406134,0.005645765,0.0027717655,0.0033248365,0.012621823],"category_scores_gemma":[0.42973307,0.00073209614,0.0012100281,0.0017855002,0.002884369,0.0064605353,0.005078615,0.0051894593,0.0019916357],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021290306,0.0008630394,0.037914112,0.017340554,0.0008082831,0.0055901627,0.13303183,0.0033379272,0.00864169,0.010332921,0.20211917,0.57789123],"study_design_scores_gemma":[0.00061079016,0.0028701718,0.09701796,0.03914076,0.0009043739,0.0064898995,0.13802211,0.005346638,0.008899985,0.023691751,0.6762576,0.0007479719],"about_ca_topic_score_codex":0.0033372447,"about_ca_topic_score_gemma":0.0035499898,"teacher_disagreement_score":0.9018675,"about_ca_system_score_codex":0.009389694,"about_ca_system_score_gemma":0.015165273,"threshold_uncertainty_score":0.5189804},"labels":[],"label_agreement":null},{"id":"W2536295050","doi":"10.1109/fskd.2016.7603424","title":"Chinese term extraction from web pages based on expected point-wise mutual information","year":2016,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Term (time); Lexicon; Point (geometry); Computer science; Information retrieval; Word (group theory); Mutual information; Information extraction; Artificial intelligence; Natural language processing; Precision and recall; Data mining; Mathematics","score_opus":0.006995051740656491,"score_gpt":0.26561186888879185,"score_spread":0.25861681714813534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2536295050","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.122727044,0.0028113222,0.8564125,0.00030746675,0.00012198754,0.0005044491,0.0048400243,0.006224059,0.006051239],"genre_scores_gemma":[0.4868327,0.0014310278,0.49612167,0.00011487407,0.00024050065,0.00067996926,0.009874689,0.00037216698,0.004332436],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9976495,0.00024638264,0.00041321135,0.0006308152,0.000883033,0.00017704925],"domain_scores_gemma":[0.9973605,0.0009930026,0.0003932268,0.00027712356,0.00089855696,0.000077548524],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011707926,0.0011101702,0.001443758,0.011414426,0.0008347354,0.001327083,0.0010388591,0.0006905856,0.0019813953],"category_scores_gemma":[0.006329372,0.0004354045,0.0017230387,0.007967515,0.0004995141,0.0026279993,0.0012554941,0.00063023594,0.0016312684],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006839393,0.0001592926,0.015542959,0.0010824554,0.00043166545,0.0012165011,0.00074435526,0.013822097,0.060524713,0.009368749,0.0102396635,0.88618356],"study_design_scores_gemma":[0.00013032948,0.0005166667,0.055979278,0.00018645333,0.0009906632,0.004165007,0.0004356738,0.7707341,0.1112919,0.030513586,0.024676567,0.00037975833],"about_ca_topic_score_codex":0.0033286647,"about_ca_topic_score_gemma":0.0039549773,"teacher_disagreement_score":0.011414426,"about_ca_system_score_codex":0.0007014879,"about_ca_system_score_gemma":0.0017425134,"threshold_uncertainty_score":0.0066284537},"labels":[],"label_agreement":null},{"id":"W2547897442","doi":"10.1017/s0140525x15001302","title":"But is it social? How to tell when groups are more than the sum of their members","year":2016,"lang":"en","type":"article","venue":"Behavioral and Brain Sciences","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Simon Fraser University","funders":"","keywords":"Perspective (graphical); Cognition; Function (biology); Psychology; Group (periodic table); Social group; Social cognition; Social psychology; Computer science; Artificial intelligence; Biology; Neuroscience","score_opus":0.06809330254961272,"score_gpt":0.33284239905683144,"score_spread":0.26474909650721873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2547897442","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20655243,0.01619345,0.5074384,0.16738321,0.0046236017,0.00030805368,0.0022081523,0.001490537,0.09380213],"genre_scores_gemma":[0.8731093,0.0050564725,0.10689308,0.006140422,0.0019646427,0.00024911994,0.00055543176,0.00034426156,0.005687287],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99441576,0.0027298862,0.00041148256,0.0011967433,0.0010217766,0.00022436444],"domain_scores_gemma":[0.95252895,0.029174995,0.0070152264,0.0059181093,0.003851821,0.0015108841],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013091993,0.0005119945,0.0010471023,0.002375861,0.0023960606,0.0062607047,0.0018764472,0.0022780653,0.006839378],"category_scores_gemma":[0.08064893,0.0003899929,0.0006286998,0.0025949355,0.013884362,0.021950576,0.0028788229,0.0028590192,0.002214993],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034878857,0.00016664117,0.05293869,0.0014821669,0.00043911528,0.00028836483,0.03461616,0.0021705625,0.0060102744,0.4305141,0.03539489,0.43563023],"study_design_scores_gemma":[0.000020383462,0.00009277059,0.0186632,0.00032670543,0.00008811743,0.0002828379,0.017212585,0.0041086026,0.0020610949,0.9279096,0.029111331,0.0001227385],"about_ca_topic_score_codex":0.0032692312,"about_ca_topic_score_gemma":0.003563579,"teacher_disagreement_score":0.013091993,"about_ca_system_score_codex":0.0011096215,"about_ca_system_score_gemma":0.0014636172,"threshold_uncertainty_score":0.06923795},"labels":[],"label_agreement":null},{"id":"W2551248486","doi":"10.5539/jmr.v8n6p105","title":"Discovery of Similarity and Dissimilarity","year":2016,"lang":"en","type":"article","venue":"Journal of Mathematics Research","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Span (engineering); Life span; Biology; Structural engineering","score_opus":0.09644546394271507,"score_gpt":0.4204893969750795,"score_spread":0.3240439330323644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2551248486","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40659148,0.0123224305,0.46058783,0.0040811608,0.0026830717,0.0010007116,0.034090843,0.008316676,0.0703258],"genre_scores_gemma":[0.77880585,0.0023111491,0.17907827,0.00037127396,0.00066006463,0.00065455516,0.027369745,0.0009905134,0.009758533],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9961511,0.00057305064,0.00039738067,0.0015048581,0.0011287255,0.0002449497],"domain_scores_gemma":[0.99435955,0.0018930818,0.0005682754,0.0015853967,0.0011065128,0.0004872157],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019552305,0.0010227424,0.0016080139,0.010146731,0.0015231371,0.0037855369,0.002261282,0.0016947476,0.014528218],"category_scores_gemma":[0.020417927,0.00052044337,0.0025844038,0.00811879,0.0015478138,0.005236047,0.0048817536,0.0021066563,0.0082589695],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017080051,0.00042810265,0.05324631,0.0029025497,0.0011620007,0.0017806804,0.0042535868,0.0032046086,0.029199673,0.09008146,0.059119564,0.7529134],"study_design_scores_gemma":[0.00036812195,0.0010561834,0.09368493,0.0014027675,0.001646974,0.007337847,0.011605376,0.13334394,0.019172229,0.4174937,0.31243062,0.00045730907],"about_ca_topic_score_codex":0.002053254,"about_ca_topic_score_gemma":0.0022210313,"teacher_disagreement_score":0.014528218,"about_ca_system_score_codex":0.00079605903,"about_ca_system_score_gemma":0.001840256,"threshold_uncertainty_score":0.048601747},"labels":[],"label_agreement":null},{"id":"W2561541699","doi":"10.1111/isj.12131","title":"Minimum sample size estimation in PLS‐SEM: The inverse square root and gamma‐exponential methods","year":2016,"lang":"en","type":"article","venue":"Information Systems Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2185,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Square root; Mathematics; Inverse; Statistics; Sample size determination; Multivariate statistics; Monte Carlo method; Exponential function; Applied mathematics; Mean squared error; Mathematical optimization; Mathematical analysis","score_opus":0.017003286307939583,"score_gpt":0.31361786399664715,"score_spread":0.2966145776887076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2561541699","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009443181,0.00023501094,0.98873186,0.0002518641,0.0000383202,0.00019960951,0.00006390877,0.00020812338,0.0008281395],"genre_scores_gemma":[0.16319527,0.0003244158,0.8342713,0.0001498971,0.00003707267,0.0008451158,0.00016842986,0.00023722921,0.00077118626],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.96176726,0.03242719,0.00072337646,0.001646958,0.0032145868,0.00022067112],"domain_scores_gemma":[0.86056924,0.12097595,0.004057577,0.0075960644,0.0063940664,0.00040708616],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.045498185,0.0014293226,0.0014555556,0.0018408947,0.00086947053,0.0021370954,0.0020672253,0.0019464265,0.0042420444],"category_scores_gemma":[0.17000197,0.00074458896,0.0014784709,0.0021695606,0.0025413008,0.0031420847,0.0025904125,0.0030571262,0.0010653894],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009782168,0.00039006968,0.010595859,0.0013580244,0.00060841447,0.00033418313,0.0022654007,0.18350534,0.006186608,0.18322913,0.0077737854,0.602775],"study_design_scores_gemma":[0.00020758754,0.00045183388,0.0052223797,0.00032941374,0.00016382139,0.00030352312,0.00041779075,0.761536,0.0068484293,0.2165294,0.007823014,0.0001667833],"about_ca_topic_score_codex":0.0017517554,"about_ca_topic_score_gemma":0.0022488504,"teacher_disagreement_score":0.045498185,"about_ca_system_score_codex":0.0010777125,"about_ca_system_score_gemma":0.0019795303,"threshold_uncertainty_score":0.24062026},"labels":[],"label_agreement":null},{"id":"W2572433599","doi":"","title":"Weak Links and Strong Meaning: The Complex Phenomenon of Negational Citations","year":2016,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Phenomenon; Meaning (existential); Computer science; Epistemology; Philosophy","score_opus":0.02508943377444567,"score_gpt":0.26274232746240656,"score_spread":0.2376528936879609,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2572433599","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27474588,0.0074419808,0.4140679,0.06085919,0.0021857393,0.0002489518,0.0014990281,0.0011229267,0.2378284],"genre_scores_gemma":[0.96309495,0.0014785102,0.024563584,0.0015669939,0.0016934868,0.0001378808,0.00036856785,0.00034230977,0.0067537827],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9897684,0.004776683,0.0007588248,0.0017157701,0.0026369554,0.000343422],"domain_scores_gemma":[0.8740761,0.10454735,0.0054429616,0.0075935028,0.0066541936,0.0016858983],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.008614249,0.00048436935,0.0010585726,0.0050189584,0.0040988214,0.011355723,0.0018745217,0.0047387388,0.015199981],"category_scores_gemma":[0.094626926,0.00077744457,0.00056748337,0.0073559694,0.010872801,0.036384054,0.006331954,0.0036615506,0.0013544327],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006583183,0.000020658403,0.0010534277,0.00027890227,0.000021895288,0.00041265733,0.0068781474,0.00034211128,0.0016152859,0.9636603,0.0033145538,0.022336176],"study_design_scores_gemma":[0.000021920137,0.000013680887,0.00065802544,0.00007756185,0.000019889234,0.00037131435,0.0012242664,0.0022070298,0.00047670474,0.9816668,0.0132336635,0.000029031573],"about_ca_topic_score_codex":0.00079032226,"about_ca_topic_score_gemma":0.00076249446,"teacher_disagreement_score":0.99498105,"about_ca_system_score_codex":0.0014939046,"about_ca_system_score_gemma":0.0012603813,"threshold_uncertainty_score":0.05084902},"labels":[],"label_agreement":null},{"id":"W2574545922","doi":"","title":"Extracting Discriminative Keyphrases with Learned Semantic Hierarchies.","year":2016,"lang":"en","type":"article","venue":"NPARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Discriminative model; Information retrieval; Salient; Set (abstract data type); Key (lock); Natural language processing; Artificial intelligence","score_opus":0.020903224723474022,"score_gpt":0.2792802311944043,"score_spread":0.2583770064709303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2574545922","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06099449,0.002768786,0.91825026,0.00057150587,0.00021725374,0.00039222857,0.002400067,0.008744115,0.0056613656],"genre_scores_gemma":[0.25225934,0.0012470064,0.7350722,0.00023351231,0.00024804566,0.00021422247,0.0052750288,0.0004494582,0.005001251],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991393,0.00015417607,0.00008037808,0.00028714034,0.00022265856,0.00011636733],"domain_scores_gemma":[0.9972633,0.0012349267,0.0004124842,0.00043842706,0.0005250301,0.00012577682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092494715,0.0013865115,0.00074917154,0.0059205694,0.00058822206,0.0012754112,0.00070818514,0.001106751,0.0030994832],"category_scores_gemma":[0.0051531605,0.00041754197,0.0009235858,0.004188735,0.0007407643,0.0032386035,0.0009983653,0.0012503172,0.0041143387],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005112207,0.00019184989,0.0025610398,0.000764512,0.00008954718,0.0005191797,0.000640877,0.005204776,0.15320092,0.00923858,0.015122704,0.8119549],"study_design_scores_gemma":[0.00023553618,0.00073588465,0.013211479,0.00025329384,0.00033016753,0.0028961431,0.0019319934,0.6399764,0.16597012,0.078023165,0.096266076,0.000169854],"about_ca_topic_score_codex":0.002432324,"about_ca_topic_score_gemma":0.0046812296,"teacher_disagreement_score":0.0059205694,"about_ca_system_score_codex":0.0007671447,"about_ca_system_score_gemma":0.0011014584,"threshold_uncertainty_score":0.010368764},"labels":[],"label_agreement":null},{"id":"W2575085358","doi":"","title":"The Quantum Chess Story.","year":2016,"lang":"en","type":"article","venue":"International journal of unconventional computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Hollywood; Quantum; Art; Aesthetics; Visual arts; Art history; Physics; Quantum mechanics","score_opus":0.013343688286291164,"score_gpt":0.30244636608169106,"score_spread":0.2891026777953999,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2575085358","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005849117,0.006832405,0.006133483,0.1062488,0.010070107,0.00004107471,0.00073504046,0.00047938176,0.86361057],"genre_scores_gemma":[0.1894475,0.0044509866,0.004306132,0.037635114,0.00340324,0.00006633466,0.0006185766,0.00068204483,0.75939],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99924004,0.00021799756,0.000022465267,0.000104037965,0.0003006393,0.00011474755],"domain_scores_gemma":[0.998963,0.0003848639,0.000047938483,0.00009186583,0.00023983602,0.0002724463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007934758,0.00042391528,0.00024076794,0.0005488809,0.006228112,0.005105679,0.00051864894,0.0022635271,0.05020551],"category_scores_gemma":[0.0060718046,0.00025201077,0.0002565222,0.0006464438,0.004446868,0.005741502,0.0022003937,0.003382002,0.01157655],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003429993,0.000009715811,0.00017885961,0.000035860077,0.0000043153304,0.0001341403,0.0015945353,0.000093254486,0.00016559835,0.3941404,0.58170927,0.021899844],"study_design_scores_gemma":[0.00000678102,0.000011351697,0.00025661776,0.000049461083,0.000001886422,0.00016234501,0.0008809193,0.00022585463,0.00023159239,0.06677381,0.9313894,0.000010028256],"about_ca_topic_score_codex":0.00795774,"about_ca_topic_score_gemma":0.011682506,"teacher_disagreement_score":0.05020551,"about_ca_system_score_codex":0.0018478314,"about_ca_system_score_gemma":0.0009912807,"threshold_uncertainty_score":0.16795415},"labels":[],"label_agreement":null},{"id":"W2576201175","doi":"","title":"Discovering Relevant Hashtags for Health Concepts: A Case Study of Twitter","year":2016,"lang":"en","type":"article","venue":"National Conference on Artificial Intelligence","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Search engine indexing; Baseline (sea); Cluster analysis; Information retrieval; Social media; Word (group theory); Natural language processing; Artificial intelligence; Data science; World Wide Web; Linguistics","score_opus":0.19415731578588039,"score_gpt":0.45427406322844993,"score_spread":0.2601167474425695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2576201175","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96701384,0.0010525599,0.016529612,0.0039962293,0.000102632526,0.00025090753,0.0063475333,0.00028847597,0.0044182725],"genre_scores_gemma":[0.97511476,0.0005464258,0.017551957,0.0004241671,0.00009432441,0.00010352102,0.0034661996,0.000046004694,0.0026527036],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99928564,0.00034231387,0.000054458753,0.00011084746,0.00014733817,0.000059411697],"domain_scores_gemma":[0.99552804,0.0035921196,0.00024078188,0.00017975076,0.00028484577,0.00017445184],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010434486,0.0005974857,0.00035564147,0.0012133961,0.000900375,0.00069827924,0.0005926235,0.0016008994,0.0017942914],"category_scores_gemma":[0.0067829676,0.000120627315,0.00043241007,0.0013506319,0.00051179837,0.0016999721,0.0005724159,0.00062244636,0.00068821444],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0040174006,0.002819116,0.5541112,0.0026498896,0.000522812,0.025416981,0.012356503,0.02868693,0.03995613,0.007813854,0.04474626,0.27690303],"study_design_scores_gemma":[0.00076347805,0.0033195224,0.3122294,0.00044305337,0.0006552455,0.019316291,0.04603435,0.40171325,0.057003513,0.01813787,0.14006706,0.00031712477],"about_ca_topic_score_codex":0.007676357,"about_ca_topic_score_gemma":0.015414483,"teacher_disagreement_score":0.007676357,"about_ca_system_score_codex":0.0006556788,"about_ca_system_score_gemma":0.0004828962,"threshold_uncertainty_score":0.015263379},"labels":[],"label_agreement":null},{"id":"W2576351195","doi":"","title":"Text Classification of Student Self-Explanations in College Physics Questions.","year":2016,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dawson College; John Abbott College; Polytechnique Montréal","funders":"","keywords":"Mathematics education; Computer science; Physics; Data science; Psychology","score_opus":0.013873316587529461,"score_gpt":0.27014708374965,"score_spread":0.2562737671621206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2576351195","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8643199,0.0026409407,0.0182725,0.0014436742,0.00041093258,0.00091543404,0.09511817,0.0036939979,0.0131843565],"genre_scores_gemma":[0.8533659,0.0005035009,0.023219384,0.00020888838,0.0002591504,0.00051708997,0.11187763,0.0002256525,0.0098228585],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980392,0.0007129674,0.00024255597,0.00032451263,0.0005762042,0.00010453156],"domain_scores_gemma":[0.96522886,0.027271925,0.0024720447,0.0008594621,0.0035077021,0.00065998686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016871989,0.00039012308,0.00029406365,0.005482196,0.0004167731,0.00091571716,0.00055035664,0.000805988,0.00905203],"category_scores_gemma":[0.023648314,0.00008092125,0.0003179286,0.002893597,0.00020810086,0.0010713653,0.00078186137,0.000502661,0.0023066062],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014517457,0.0010065592,0.22317457,0.003828764,0.00030307533,0.00075930107,0.004037056,0.0025977767,0.028742194,0.0022641667,0.11523184,0.61660296],"study_design_scores_gemma":[0.00021736315,0.0007596878,0.6910276,0.000986143,0.00033949234,0.0009243667,0.0055197645,0.09057577,0.03870865,0.004124732,0.16671076,0.00010565083],"about_ca_topic_score_codex":0.0021561084,"about_ca_topic_score_gemma":0.004747085,"teacher_disagreement_score":0.00905203,"about_ca_system_score_codex":0.0006324135,"about_ca_system_score_gemma":0.00069009315,"threshold_uncertainty_score":0.03028208},"labels":[],"label_agreement":null},{"id":"W2580236052","doi":"10.29173/cais647","title":"Tags, Homonyms, and the Manifestation of Intentionality","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Intentionality; Recall; Relevance (law); Humanities; Philosophy; Psychology; Epistemology; Cognitive psychology; Political science","score_opus":0.022850301315147908,"score_gpt":0.25644568886519725,"score_spread":0.23359538755004935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2580236052","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4848787,0.00467579,0.3653653,0.009366506,0.00053701375,0.0002126531,0.00045677891,0.0005102251,0.13399698],"genre_scores_gemma":[0.96596724,0.000643605,0.02796901,0.00038920346,0.0000959907,0.00006367237,0.00021179844,0.00010826739,0.0045512994],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9948572,0.0028788336,0.00035425677,0.0006876891,0.0009477619,0.00027420567],"domain_scores_gemma":[0.9825012,0.011902716,0.0014003036,0.0029202788,0.0009647081,0.0003107721],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005067597,0.00032484502,0.00045956165,0.0032317624,0.0022656193,0.007553015,0.00081237475,0.0013987308,0.0033231354],"category_scores_gemma":[0.024279242,0.00052722683,0.00050328835,0.0022111235,0.01354823,0.015869264,0.0038686227,0.002023426,0.00045602134],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016433901,0.00004252643,0.0067483196,0.00034227778,0.000040186875,0.00081456034,0.07958322,0.00044183646,0.0073570204,0.84255445,0.0019311226,0.059980176],"study_design_scores_gemma":[0.00003425343,0.0000707095,0.007874454,0.0003450225,0.00005192406,0.0021049678,0.03639569,0.004251343,0.005815621,0.88816404,0.05478845,0.000103515136],"about_ca_topic_score_codex":0.0014257481,"about_ca_topic_score_gemma":0.0014908705,"teacher_disagreement_score":0.007553015,"about_ca_system_score_codex":0.0015931589,"about_ca_system_score_gemma":0.0012651545,"threshold_uncertainty_score":0.026800334},"labels":[],"label_agreement":null},{"id":"W2581505073","doi":"10.1109/icdim.2016.7829792","title":"Extracting keyword and keyphrase from online privacy policies","year":2016,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Computer science; Taxonomy (biology); Keyword extraction; Domain (mathematical analysis); Annotation; Artificial intelligence; Natural language processing; Rake; Information retrieval; Variety (cybernetics); Key (lock)","score_opus":0.021159089162077555,"score_gpt":0.2985420800903012,"score_spread":0.27738299092822366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2581505073","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2975919,0.0050257845,0.61023474,0.0021278034,0.00041985692,0.001417881,0.05140312,0.014580052,0.017198946],"genre_scores_gemma":[0.3751046,0.002324792,0.5835956,0.00014035527,0.00010779089,0.0004788808,0.032962095,0.00058707735,0.004698861],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996768,0.00070415164,0.000609743,0.0005793867,0.00109243,0.0002462807],"domain_scores_gemma":[0.9864225,0.006599369,0.0015048303,0.0019899737,0.0032253338,0.0002581211],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002474102,0.00088176836,0.00088430796,0.00975358,0.001080868,0.0020967068,0.0005838633,0.001144142,0.00203283],"category_scores_gemma":[0.019193336,0.0004821971,0.00082425115,0.007735523,0.00072847446,0.00616082,0.0011952313,0.0014245964,0.003065248],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005801869,0.00028633082,0.017744504,0.0034804146,0.0000933716,0.0017777508,0.0052832146,0.009219301,0.09281394,0.025226936,0.025702864,0.8177912],"study_design_scores_gemma":[0.0001412501,0.00057131547,0.041934337,0.000919883,0.00025720234,0.0057686567,0.01486228,0.15064809,0.25755557,0.07030077,0.45660654,0.00043417196],"about_ca_topic_score_codex":0.0060108704,"about_ca_topic_score_gemma":0.0074763796,"teacher_disagreement_score":0.00975358,"about_ca_system_score_codex":0.0018489681,"about_ca_system_score_gemma":0.004626913,"threshold_uncertainty_score":0.013415277},"labels":[],"label_agreement":null},{"id":"W2586790131","doi":"10.4236/jilsa.2017.91002","title":"Text-Based Intelligent Learning Emotion System","year":2017,"lang":"en","type":"article","venue":"Journal of Intelligent Learning Systems and Applications","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Meaning (existential); Feeling; Experiential learning; Social media; Emotion recognition; Human–computer interaction; Artificial intelligence; World Wide Web; Psychology; Social psychology","score_opus":0.02001367212382593,"score_gpt":0.297721264805105,"score_spread":0.2777075926812791,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2586790131","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2497368,0.0024527856,0.6430269,0.0016997486,0.0013727861,0.0017760042,0.015379039,0.06392245,0.020633511],"genre_scores_gemma":[0.6210604,0.0009503132,0.32331997,0.00077655155,0.00051646674,0.0014208099,0.026380904,0.0003793058,0.025195275],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999463,0.00007626286,0.00008269,0.00018308891,0.000145221,0.00004976965],"domain_scores_gemma":[0.9994073,0.00011531597,0.00005329571,0.00005143678,0.00033606373,0.00003656723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00060265017,0.0008238249,0.0007760412,0.0011970355,0.00037120827,0.0008541633,0.0010533003,0.0007617115,0.00650312],"category_scores_gemma":[0.0017231208,0.00012366656,0.0006082446,0.0006308047,0.00013741804,0.0014660235,0.0005339345,0.0007179915,0.005652914],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014180047,0.0009237099,0.00611439,0.0004260933,0.0001564809,0.0004298289,0.0002399121,0.008728341,0.09610896,0.0016557022,0.04049056,0.8433081],"study_design_scores_gemma":[0.0001562778,0.00065681996,0.014348593,0.000055506647,0.0002399119,0.00043087837,0.00030890122,0.8562698,0.09543753,0.004308675,0.027695699,0.00009133077],"about_ca_topic_score_codex":0.0011567958,"about_ca_topic_score_gemma":0.0011382787,"teacher_disagreement_score":0.00650312,"about_ca_system_score_codex":0.0005451358,"about_ca_system_score_gemma":0.00023632974,"threshold_uncertainty_score":0.021755159},"labels":[],"label_agreement":null},{"id":"W2587261082","doi":"10.29173/cais456","title":"Adaptation of a Key Phrase Extractor for Japanese Text","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Extractor; Phrase; Key (lock); Adaptation (eye); Natural language processing; Artificial intelligence; Information retrieval","score_opus":0.024650770768542413,"score_gpt":0.2595055367525164,"score_spread":0.234854765983974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2587261082","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013309413,0.0005249511,0.92438215,0.00021606461,0.00047406976,0.00054056343,0.0028241412,0.054874267,0.002854344],"genre_scores_gemma":[0.03124595,0.00043575754,0.9477049,0.00015620628,0.0001826146,0.00039719295,0.0065727243,0.0038444027,0.009460337],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99867165,0.00014639074,0.00021944504,0.00046309634,0.00040184293,0.00009767512],"domain_scores_gemma":[0.9974617,0.0005269691,0.00013214997,0.00057252747,0.0011798056,0.00012685238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013591574,0.0019815944,0.0013741087,0.003040164,0.0009504035,0.0015197629,0.0014209821,0.0008262281,0.0138698695],"category_scores_gemma":[0.004327321,0.00092697283,0.0011877873,0.0031692688,0.0004389581,0.003170604,0.0014778416,0.0017445325,0.013997972],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005761605,0.00014853012,0.0017123362,0.0008066768,0.00018709582,0.0005891368,0.00043267402,0.0014390743,0.18359324,0.0024435546,0.023151062,0.7849205],"study_design_scores_gemma":[0.00036302223,0.0005555694,0.012540276,0.00014251366,0.00058697426,0.0029476539,0.00069622137,0.18841578,0.38505667,0.0037906202,0.40441445,0.0004902413],"about_ca_topic_score_codex":0.0057111736,"about_ca_topic_score_gemma":0.009134576,"teacher_disagreement_score":0.0138698695,"about_ca_system_score_codex":0.0005266869,"about_ca_system_score_gemma":0.001448043,"threshold_uncertainty_score":0.046399355},"labels":[],"label_agreement":null},{"id":"W2587364723","doi":"10.29173/cais440","title":"In Search of the Perfect Filter: Indexing Theory Implications for Internet Blocking and Rating Software Products","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Blocking (statistics); The Internet; Legislature; Search engine indexing; Globe; Politics; Government (linguistics); Software; Control (management); Filter (signal processing); Internet access; Business; Public relations; Political science; Computer science; World Wide Web; Artificial intelligence","score_opus":0.02440889892602804,"score_gpt":0.2713012996301007,"score_spread":0.24689240070407265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2587364723","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14219862,0.0036170636,0.76437616,0.027359081,0.00040870538,0.00023217268,0.0010139629,0.00066388637,0.06013026],"genre_scores_gemma":[0.9221798,0.0015343798,0.059728075,0.0022863871,0.0009860279,0.00021308775,0.0005240449,0.0001916087,0.012356614],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9908057,0.004212659,0.0005372166,0.0019445084,0.001789527,0.0007103819],"domain_scores_gemma":[0.78723055,0.17702676,0.012411014,0.012195061,0.008741026,0.0023956054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023117715,0.0005873882,0.002586413,0.0050201113,0.0035211574,0.009454764,0.0035192121,0.0042755483,0.01944457],"category_scores_gemma":[0.18125497,0.00090208213,0.001385451,0.005541594,0.007810895,0.0265276,0.0033853916,0.0041958294,0.002090978],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014052984,0.000095984775,0.005606289,0.00012734217,0.000060849747,0.000091235015,0.000694589,0.0043289666,0.00024669987,0.9491015,0.00488861,0.034617517],"study_design_scores_gemma":[0.00006344602,0.00003516486,0.0014409955,0.000037517748,0.000026042788,0.00011743472,0.00020809639,0.047115635,0.00019702772,0.9488721,0.0018621177,0.000024456702],"about_ca_topic_score_codex":0.013986256,"about_ca_topic_score_gemma":0.008358138,"teacher_disagreement_score":0.023117715,"about_ca_system_score_codex":0.0038775601,"about_ca_system_score_gemma":0.0033535315,"threshold_uncertainty_score":0.12225968},"labels":[],"label_agreement":null},{"id":"W2588009921","doi":"10.29173/cais305","title":"Deriving an Ontology of Reader Authored Markings Made on Electronic Documents","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Reading (process); Ontology; Information retrieval; World Wide Web; Humanities; Library science; Art; Linguistics; Philosophy; Epistemology","score_opus":0.024635255279582817,"score_gpt":0.2773505347787903,"score_spread":0.25271527949920747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2588009921","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21988575,0.00075962604,0.71777266,0.0018609712,0.0002051711,0.0005875098,0.0008606373,0.0009355972,0.057132065],"genre_scores_gemma":[0.62911874,0.0006728451,0.35652098,0.00019712985,0.000069623275,0.0003491913,0.0013458661,0.00036220267,0.011363416],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99312055,0.002603343,0.00079580885,0.0013337831,0.001706326,0.0004401978],"domain_scores_gemma":[0.9873395,0.0057944627,0.001514669,0.0026465885,0.002287781,0.0004170022],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.005763986,0.00046451337,0.00044877693,0.005847127,0.0027588264,0.0065948563,0.001509847,0.0015760672,0.0029887923],"category_scores_gemma":[0.0166061,0.00083845784,0.0013893361,0.0038508156,0.005560272,0.012032288,0.002913703,0.0020342283,0.0007523542],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001508777,0.00028053447,0.032183588,0.00066509354,0.000085999345,0.0022917604,0.10554578,0.0057433154,0.019984366,0.689267,0.002650432,0.14115134],"study_design_scores_gemma":[0.00007012794,0.00023829963,0.047549143,0.0011815407,0.00028964315,0.0035255752,0.09147942,0.07398291,0.030530326,0.44767454,0.30316573,0.0003127688],"about_ca_topic_score_codex":0.010716821,"about_ca_topic_score_gemma":0.011381296,"teacher_disagreement_score":0.99340516,"about_ca_system_score_codex":0.0036666023,"about_ca_system_score_gemma":0.0045859944,"threshold_uncertainty_score":0.030483246},"labels":[],"label_agreement":null},{"id":"W2588137444","doi":"10.29173/cais356","title":"Integrating Knowledge from Different Sources for Automatic Back-of-the-book Indexing","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Indexation; Search engine indexing; Computer science; Valuation (finance); Humanities; Information retrieval; Library science; Art; Business","score_opus":0.026891346411433786,"score_gpt":0.2663705937485865,"score_spread":0.23947924733715273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2588137444","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0756041,0.00641689,0.84145087,0.0016053014,0.0003731087,0.0017411022,0.005632594,0.010181137,0.056994848],"genre_scores_gemma":[0.18201227,0.003255935,0.7930004,0.00026984556,0.0002584799,0.00058011577,0.010002585,0.0013879183,0.009232577],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9939214,0.0018140691,0.00063949655,0.0010251753,0.0022889269,0.00031099535],"domain_scores_gemma":[0.97345304,0.015842965,0.0010196237,0.0045192726,0.0047239456,0.00044121864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056144325,0.0011895705,0.001704122,0.028162835,0.0019046474,0.008123061,0.002202568,0.0017681202,0.011013378],"category_scores_gemma":[0.03522473,0.0008058068,0.001666607,0.019759357,0.0013134837,0.011000262,0.0054605124,0.0017160485,0.0065603796],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026996262,0.00027496365,0.00435697,0.001093961,0.000223285,0.00040634398,0.0019404769,0.0021014279,0.008428653,0.00721752,0.005614522,0.9680721],"study_design_scores_gemma":[0.0005572875,0.00096232904,0.044981796,0.0030246482,0.002510019,0.0040237033,0.013065756,0.19769673,0.17831895,0.16260122,0.39122453,0.0010330956],"about_ca_topic_score_codex":0.0061608027,"about_ca_topic_score_gemma":0.010258943,"teacher_disagreement_score":0.028162835,"about_ca_system_score_codex":0.0018061065,"about_ca_system_score_gemma":0.0037698648,"threshold_uncertainty_score":0.03684348},"labels":[],"label_agreement":null},{"id":"W2588474510","doi":"10.29173/cais596","title":"Facets of Serendipity in Everyday Chance Encounters: Content Analysis of Social Media Accounts","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Serendipity; Humanities; Sociology; Art; Epistemology; Philosophy","score_opus":0.05236118604178577,"score_gpt":0.2774246457792693,"score_spread":0.2250634597374835,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2588474510","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97405833,0.0005739875,0.012122979,0.00044939804,0.00004199588,0.00012437026,0.00095403695,0.00011379197,0.011561035],"genre_scores_gemma":[0.99463403,0.0002859699,0.0032214508,0.000032071715,0.000031486445,0.00009948292,0.00045117896,0.000050472645,0.001193897],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9969265,0.0017183094,0.00021057743,0.00030919805,0.00064373453,0.00019170002],"domain_scores_gemma":[0.9767568,0.018452138,0.0024049326,0.0010929818,0.0008351145,0.00045813684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002251556,0.00033518934,0.00033056468,0.0053416598,0.0020941375,0.00404371,0.00051016966,0.0006536374,0.0018563485],"category_scores_gemma":[0.019448651,0.00032625534,0.00039694033,0.005378899,0.0029532954,0.0054850434,0.002790031,0.0007107015,0.00026146788],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042213337,0.00005627388,0.14851272,0.0008112767,0.00010251788,0.0014195045,0.7398587,0.0010519718,0.0065074717,0.022178747,0.0027103438,0.07636832],"study_design_scores_gemma":[0.000021122436,0.000101613725,0.30510217,0.0008270956,0.000103778766,0.0024369254,0.58044714,0.010275241,0.0038626012,0.016708698,0.079884864,0.00022878597],"about_ca_topic_score_codex":0.0044569382,"about_ca_topic_score_gemma":0.0070834244,"teacher_disagreement_score":0.0053416598,"about_ca_system_score_codex":0.001398936,"about_ca_system_score_gemma":0.00076166866,"threshold_uncertainty_score":0.011907458},"labels":[],"label_agreement":null},{"id":"W2588863515","doi":"10.29173/cais505","title":"Assessing Intra and Extra Web-based Automatic Indexing Tools","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Centre de Développement du Porc du Québec; University of Toronto","funders":"","keywords":"Search engine indexing; Extractor; Information retrieval; Computer science; Representation (politics); Automatic indexing; World Wide Web; Engineering","score_opus":0.027889095759107196,"score_gpt":0.2721698050000837,"score_spread":0.2442807092409765,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2588863515","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8586821,0.0021106035,0.10672901,0.00044697413,0.00023175968,0.0013586192,0.0011381032,0.006171382,0.023131441],"genre_scores_gemma":[0.86027694,0.00057853747,0.12925589,0.00023473945,0.00022192448,0.0008081024,0.0026595038,0.0012817384,0.0046826308],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9632122,0.018299814,0.003771518,0.0028460387,0.011014288,0.0008561962],"domain_scores_gemma":[0.71102697,0.23777443,0.009000325,0.015161373,0.025026929,0.0020099261],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.028730223,0.0010041795,0.0010162387,0.009926886,0.0007865364,0.005913639,0.0016815077,0.0013742102,0.003296365],"category_scores_gemma":[0.12555689,0.0004812415,0.0006137313,0.00452158,0.0010420564,0.010264111,0.003686189,0.001349352,0.0019907362],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003556195,0.0023913188,0.085122734,0.0021853033,0.00048614957,0.00024631715,0.004650373,0.0015858269,0.032666463,0.0019924042,0.0021237682,0.8629931],"study_design_scores_gemma":[0.0011403499,0.022644818,0.6047964,0.0013783701,0.002967324,0.0036998796,0.012079954,0.076193675,0.20218684,0.008749221,0.0633003,0.00086277013],"about_ca_topic_score_codex":0.00057584286,"about_ca_topic_score_gemma":0.0010789442,"teacher_disagreement_score":0.028730223,"about_ca_system_score_codex":0.00065838645,"about_ca_system_score_gemma":0.0010593134,"threshold_uncertainty_score":0.15194178},"labels":[],"label_agreement":null},{"id":"W25922597","doi":"10.4110/in.2015.15.2.83","title":"Using Subjective Adjectives in Opinion Retrieval from Blogs.","year":2007,"lang":"en","type":"article","venue":"Immune Network","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Task (project management); Relevance (law); Information retrieval; Point (geometry); Event (particle physics); Sentiment analysis; Product (mathematics); Subject (documents); Psychology; Natural language processing; World Wide Web; Political science; Mathematics","score_opus":0.025391575791442293,"score_gpt":0.32161483975794497,"score_spread":0.29622326396650267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W25922597","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7049066,0.007689669,0.10315637,0.0035385008,0.0028866907,0.0047457335,0.03529212,0.0043971594,0.1333872],"genre_scores_gemma":[0.9279442,0.0012481423,0.0456996,0.00074818823,0.00066836947,0.0017336748,0.009451003,0.00022809763,0.012278743],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9931399,0.0034046988,0.00065832766,0.0007494561,0.001757383,0.00029010302],"domain_scores_gemma":[0.9760039,0.016365396,0.002225857,0.0007311718,0.003997101,0.00067663885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053669782,0.0009564106,0.00059669354,0.004838843,0.00071084884,0.004356885,0.0005135603,0.0013726883,0.016156776],"category_scores_gemma":[0.042511526,0.00016993382,0.00078907225,0.002703479,0.0006222498,0.0043201935,0.0020491008,0.0009566101,0.0071703442],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004135618,0.00063518726,0.10487962,0.0064332867,0.0005162766,0.000724776,0.015995722,0.0018959057,0.022698509,0.0057383813,0.06910676,0.7672399],"study_design_scores_gemma":[0.00075340684,0.0037826495,0.5555092,0.00267826,0.0011282258,0.0024332334,0.07138501,0.11346896,0.018402496,0.033578206,0.19576494,0.0011153464],"about_ca_topic_score_codex":0.0012497938,"about_ca_topic_score_gemma":0.0025612388,"teacher_disagreement_score":0.016156776,"about_ca_system_score_codex":0.0010749163,"about_ca_system_score_gemma":0.00037460256,"threshold_uncertainty_score":0.05404973},"labels":[],"label_agreement":null},{"id":"W2593550342","doi":"10.1371/journal.pone.0184188","title":"Multi-level computational methods for interdisciplinary research in the HathiTrust Digital Library","year":2017,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Hospitality and Tourism Research Centre, Hong Kong Polytechnic University; National Endowment for the Humanities","keywords":"Computer science; Digital library; Argument (complex analysis); Reading (process); Information retrieval; Domain (mathematical analysis); Zoom; Set (abstract data type); Resource (disambiguation); Data science; Identification (biology); Library classification; World Wide Web; Linguistics","score_opus":0.4036673539531947,"score_gpt":0.49515156111194014,"score_spread":0.09148420715874545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2593550342","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0076874993,0.00065999303,0.9881094,0.001163036,0.00003268703,0.0000715428,0.00015458182,0.0003366224,0.0017846343],"genre_scores_gemma":[0.096069574,0.000372589,0.9005502,0.00021210722,0.00008862789,0.0005653366,0.00036780193,0.00012547668,0.0016483519],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99636334,0.00236861,0.00024222076,0.00040941624,0.00049675023,0.000119751174],"domain_scores_gemma":[0.97671336,0.019442948,0.00085169764,0.0019319908,0.0007438837,0.00031608256],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0066945287,0.00059618586,0.0011397761,0.004790693,0.00141382,0.0049301283,0.0020538373,0.0013661947,0.0042623817],"category_scores_gemma":[0.027023928,0.0007447781,0.0026766239,0.00507449,0.002297537,0.0043356004,0.004604713,0.0022516341,0.00071327045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008725263,0.00014139166,0.0034939286,0.00041641074,0.00027233118,0.00018059746,0.0015445058,0.23893735,0.0010983606,0.59876835,0.0045863558,0.15047316],"study_design_scores_gemma":[0.000013677143,0.000009924783,0.00043112593,0.000028820377,0.00001879191,0.000016685492,0.00011374508,0.6980552,0.00019189842,0.29687494,0.004232477,0.000012745546],"about_ca_topic_score_codex":0.0072186096,"about_ca_topic_score_gemma":0.014319992,"teacher_disagreement_score":0.0072186096,"about_ca_system_score_codex":0.0030646257,"about_ca_system_score_gemma":0.0025089323,"threshold_uncertainty_score":0.035404444},"labels":[],"label_agreement":null},{"id":"W2594482583","doi":"10.7202/1038906ar","title":"Analyse statistique des évangiles synoptiques : une étude de la paternité des textes par l’analyse des correspondances du taxi","year":2017,"lang":"fr","type":"article","venue":"Revue de l’Université de Moncton","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Moncton","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.017957825344564495,"score_gpt":0.2706624048675394,"score_spread":0.2527045795229749,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2594482583","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8250991,0.0025357676,0.14327058,0.0017451101,0.00016660352,0.0001911863,0.00765107,0.0007790739,0.01856148],"genre_scores_gemma":[0.9564987,0.0005059379,0.034959283,0.000102250284,0.00007032259,0.00018698849,0.002684341,0.0002906222,0.0047015673],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99539703,0.0019277399,0.00035247504,0.0010945017,0.001111162,0.00011721149],"domain_scores_gemma":[0.9368926,0.051113777,0.0041598664,0.003004225,0.004539256,0.00029022424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005127391,0.00043315432,0.00044005408,0.0051521487,0.0009502713,0.003749066,0.00067527534,0.0005574663,0.006908312],"category_scores_gemma":[0.03928821,0.00034692386,0.0006107198,0.0068073706,0.0020890625,0.0040193405,0.0009176457,0.0012145691,0.0016429495],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011807466,0.00013539927,0.3869534,0.0018688709,0.0008609338,0.0013641445,0.04516024,0.007854572,0.044944417,0.051433608,0.008354984,0.4498886],"study_design_scores_gemma":[0.00007815642,0.00032794278,0.7214419,0.00069555774,0.000424824,0.0030865395,0.02736282,0.06956042,0.030463329,0.05144512,0.09482466,0.00028861963],"about_ca_topic_score_codex":0.0062910616,"about_ca_topic_score_gemma":0.0071896543,"teacher_disagreement_score":0.006908312,"about_ca_system_score_codex":0.0011895688,"about_ca_system_score_gemma":0.0011349865,"threshold_uncertainty_score":0.027116537},"labels":[],"label_agreement":null},{"id":"W2594787554","doi":"10.19173/irrodl.v18i1.2646","title":"Social Web Content Enhancement in a Distance Learning Environment: Intelligent Metadata Generation for Resources","year":2017,"lang":"en","type":"article","venue":"The International Review of Research in Open and Distributed Learning","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Comisión de Operación y Fomento de Actividades Académicas, Instituto Politécnico Nacional; Instituto Politécnico Nacional","keywords":"Metadata; Computer science; World Wide Web; Information retrieval; Classifier (UML); Class (philosophy); Metadata repository; Multimedia; Artificial intelligence","score_opus":0.23984678601347464,"score_gpt":0.47129335270847894,"score_spread":0.2314465666950043,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2594787554","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14630988,0.00090984773,0.8388426,0.0014631723,0.00019274422,0.00039231984,0.00040588767,0.0050797304,0.0064038504],"genre_scores_gemma":[0.52075297,0.00060566043,0.4659734,0.00024828617,0.00020682564,0.00021411896,0.0010090707,0.0002553593,0.010734234],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993631,0.00019481168,0.000045022913,0.000128027,0.0002130848,0.00005594553],"domain_scores_gemma":[0.99842596,0.0007284434,0.00016430988,0.00024234194,0.00035235097,0.00008665775],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009904263,0.00039471764,0.0004896844,0.0021521233,0.00061177224,0.0011823655,0.0008800158,0.000903715,0.0015704937],"category_scores_gemma":[0.0026261853,0.00016882781,0.0005437225,0.0016863332,0.00053145393,0.0022288796,0.00071066903,0.0005750747,0.0014596921],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017943075,0.00039320212,0.005912776,0.00017398056,0.000045985787,0.00024251362,0.00032552323,0.013942752,0.027907439,0.0045349826,0.0067574014,0.939584],"study_design_scores_gemma":[0.00005356014,0.00020460844,0.008514359,0.00005025199,0.00008214642,0.00046641755,0.00043446134,0.8791151,0.07432679,0.014460578,0.022239763,0.00005194694],"about_ca_topic_score_codex":0.0023950827,"about_ca_topic_score_gemma":0.0028831598,"teacher_disagreement_score":0.0023950827,"about_ca_system_score_codex":0.0007576719,"about_ca_system_score_gemma":0.0007035899,"threshold_uncertainty_score":0.005497396},"labels":[],"label_agreement":null},{"id":"W2601411009","doi":"10.29173/cais254","title":"A Pilot Study of Enhancing Subject Discovery of Textual Web Resources","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Subject (documents); World Wide Web; Computer science; Web page; Search engine indexing; Information retrieval; Library science; Humanities; Art","score_opus":0.028707232483256913,"score_gpt":0.26272599576058425,"score_spread":0.23401876327732735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2601411009","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9944865,0.000047118978,0.002171929,0.000098491386,0.000017210465,0.001088739,0.00020971335,0.00020187833,0.0016784687],"genre_scores_gemma":[0.96972704,0.00013740182,0.022731878,0.00024402495,0.000057187972,0.0015229437,0.0005491779,0.00007890021,0.0049514635],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9979664,0.0012072347,0.00017241995,0.00026336574,0.000239124,0.00015148737],"domain_scores_gemma":[0.95475477,0.036747314,0.0012316884,0.0029650643,0.0026622072,0.001638921],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047474047,0.0005630122,0.0006810715,0.0007355206,0.0007259519,0.0013119533,0.0011578287,0.0009615215,0.0073400987],"category_scores_gemma":[0.02541498,0.00040223377,0.000348346,0.0006827042,0.00066262006,0.002205134,0.0009413962,0.00073081284,0.002066583],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.024374725,0.1311207,0.07516761,0.0046140132,0.0002753271,0.0038882096,0.062938236,0.0036147519,0.18461856,0.0018298574,0.0057189837,0.50183904],"study_design_scores_gemma":[0.012927259,0.38935244,0.28970993,0.00040270042,0.0010469938,0.0032136142,0.043324247,0.034350898,0.16778612,0.0044848393,0.052981216,0.00041971868],"about_ca_topic_score_codex":0.0016450743,"about_ca_topic_score_gemma":0.0018554925,"teacher_disagreement_score":0.0073400987,"about_ca_system_score_codex":0.00035677332,"about_ca_system_score_gemma":0.0010180576,"threshold_uncertainty_score":0.025106966},"labels":[],"label_agreement":null},{"id":"W2603222250","doi":"10.1017/s0269888917000029","title":"The state of the art in semantic relatedness: a framework for comparison","year":2017,"lang":"en","type":"article","venue":"The Knowledge Engineering Review","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; WordNet; Closeness; Semantic similarity; Similarity (geometry); Strengths and weaknesses; Representation (politics); Domain (mathematical analysis); Metric (unit); Data science; Information retrieval; Artificial intelligence; Mathematics","score_opus":0.022776585487954894,"score_gpt":0.33508661123122313,"score_spread":0.3123100257432682,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2603222250","genre_codex":"review","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00892645,0.56005424,0.38789323,0.011939467,0.0012078224,0.0003692281,0.00069898466,0.00044149536,0.028469041],"genre_scores_gemma":[0.25858092,0.25137886,0.4767467,0.0033231883,0.003124244,0.0016415712,0.0019850268,0.00036687576,0.0028525728],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.96676606,0.019896224,0.0028401918,0.0025477193,0.007410995,0.000538854],"domain_scores_gemma":[0.9177903,0.06216825,0.0041583953,0.004887352,0.010264881,0.0007307557],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03425916,0.0014155229,0.0025559962,0.04748476,0.0016594288,0.008674193,0.0038719208,0.002807056,0.0038329987],"category_scores_gemma":[0.082191154,0.00049442786,0.0016807601,0.03777338,0.0058944346,0.021757618,0.0039170557,0.0034103293,0.0013249562],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008743895,0.00013175482,0.0026971356,0.006822943,0.00036368732,0.00009339768,0.0014202596,0.0046013864,0.0009317754,0.49641952,0.0150979115,0.47133276],"study_design_scores_gemma":[0.00004023686,0.00035222986,0.008228969,0.014673682,0.0004978082,0.0005381702,0.0042340867,0.045560416,0.0023105491,0.62946486,0.29388446,0.00021451976],"about_ca_topic_score_codex":0.0033574062,"about_ca_topic_score_gemma":0.002051172,"teacher_disagreement_score":0.04748476,"about_ca_system_score_codex":0.006035258,"about_ca_system_score_gemma":0.0037104823,"threshold_uncertainty_score":0.18118191},"labels":[],"label_agreement":null},{"id":"W2605327093","doi":"","title":"Laval University at TREC Dynamic Domain 2016: Subtopic extraction focused on Named Entities.","year":2016,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université Laval; Lakehead University","funders":"","keywords":"Computer science; Domain (mathematical analysis); Extraction (chemistry); Information extraction; Named-entity recognition; Artificial intelligence; Engineering; Systems engineering; Chemistry; Mathematics; Chromatography","score_opus":0.014616333478574932,"score_gpt":0.25083006232655075,"score_spread":0.2362137288479758,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605327093","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.087178074,0.011527807,0.0915122,0.0073640333,0.004534156,0.0030719526,0.6601518,0.06850435,0.06615562],"genre_scores_gemma":[0.058257923,0.0013106929,0.10929217,0.0007768828,0.00046453674,0.001026219,0.78111124,0.0034292343,0.044331104],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99623495,0.0010907311,0.00030297638,0.00089316384,0.0010694314,0.000408721],"domain_scores_gemma":[0.99272835,0.0015700292,0.00020563391,0.0012978442,0.003460687,0.00073744985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037793792,0.0025887217,0.0015469326,0.0066690166,0.0025399872,0.0030404415,0.0024149248,0.0022443293,0.023103498],"category_scores_gemma":[0.010994697,0.0007518034,0.0010777885,0.0034243523,0.0007530657,0.0054537184,0.0034584699,0.0025796182,0.022780355],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006122097,0.00045828367,0.0015447895,0.0010642731,0.00009556577,0.0002579203,0.00034322194,0.0013091075,0.035213552,0.0017954136,0.79675835,0.1605473],"study_design_scores_gemma":[0.0007216169,0.0004221651,0.016853888,0.00047548432,0.00022813691,0.0009621403,0.0012679925,0.032701135,0.08925302,0.0046480875,0.852206,0.00026038382],"about_ca_topic_score_codex":0.06180754,"about_ca_topic_score_gemma":0.087615386,"teacher_disagreement_score":0.06180754,"about_ca_system_score_codex":0.0023041537,"about_ca_system_score_gemma":0.006504895,"threshold_uncertainty_score":0.12289554},"labels":[],"label_agreement":null},{"id":"W2605475111","doi":"10.1007/978-3-319-57351-9_23","title":"Learning Physical Properties of Objects Using Gaussian Mixture Models","year":2017,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; WordNet; Taxonomy (biology); Artificial intelligence; Classifier (UML); Inference; Gaussian; Machine learning; Data mining","score_opus":0.033495551435880865,"score_gpt":0.27873265143962406,"score_spread":0.24523710000374319,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605475111","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008447013,0.0003987669,0.989684,0.00006637697,0.000018196293,0.000014674382,0.00006813066,0.0006542595,0.0006487256],"genre_scores_gemma":[0.41346532,0.002572509,0.5752169,0.00018642108,0.00014685761,0.0001511257,0.0014808441,0.000549522,0.006230414],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995987,0.000070579146,0.000019099842,0.00014958114,0.00013585894,0.000026293688],"domain_scores_gemma":[0.9991303,0.0004947467,0.00009283175,0.00016564224,0.00008357253,0.000032903412],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078778755,0.0011276298,0.0013858897,0.001351404,0.00031031127,0.0014953804,0.0017351328,0.0015284766,0.0016782552],"category_scores_gemma":[0.002633021,0.0010145893,0.0020854045,0.0015246086,0.00095540547,0.003329256,0.0017019064,0.002142127,0.0014365079],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016110767,0.00013865985,0.0024348714,0.00028506477,0.00024156843,0.00012493705,0.00016834373,0.40362117,0.01832483,0.019549575,0.0030717342,0.55187815],"study_design_scores_gemma":[0.0000043782798,0.000022296425,0.0006620978,0.00001437849,0.000027448412,0.00008225218,0.000017646858,0.9694119,0.0024131946,0.026302787,0.0010260382,0.000015630121],"about_ca_topic_score_codex":0.0017959337,"about_ca_topic_score_gemma":0.0019599174,"teacher_disagreement_score":0.0017959337,"about_ca_system_score_codex":0.00050089037,"about_ca_system_score_gemma":0.00036793767,"threshold_uncertainty_score":0.0056143403},"labels":[],"label_agreement":null},{"id":"W2612942342","doi":"","title":"Contextualisation de messages courts :l’importance des métadonnées","year":2013,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Political science","score_opus":0.027389263088150966,"score_gpt":0.2603885544003364,"score_spread":0.23299929131218547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2612942342","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09435086,0.0060885665,0.82951206,0.011930632,0.001794503,0.0006652592,0.0019043644,0.0034473771,0.050306436],"genre_scores_gemma":[0.80153704,0.0016639746,0.18166053,0.0011772321,0.0009533257,0.0003529084,0.0013602477,0.0014640625,0.009830667],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9790188,0.009798097,0.0016908667,0.0035716672,0.005242589,0.000677923],"domain_scores_gemma":[0.9134769,0.058282495,0.0056574605,0.009985525,0.011349169,0.0012485287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010013773,0.0010509178,0.0011121475,0.004296135,0.00255604,0.012219794,0.0017650317,0.00303458,0.009094044],"category_scores_gemma":[0.07766823,0.0010229294,0.0009322975,0.0025858716,0.003886128,0.01693324,0.0041907146,0.0036288504,0.0023201627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012463259,0.00021180874,0.018546594,0.0026606156,0.0003488348,0.001985207,0.053019583,0.00656571,0.044422187,0.4165519,0.017556619,0.43688458],"study_design_scores_gemma":[0.00016623327,0.0005163941,0.02181752,0.003309415,0.000822244,0.003148434,0.03577657,0.10135388,0.056027252,0.3588526,0.41773415,0.00047534812],"about_ca_topic_score_codex":0.0057302904,"about_ca_topic_score_gemma":0.0037658303,"teacher_disagreement_score":0.012219794,"about_ca_system_score_codex":0.0025100352,"about_ca_system_score_gemma":0.0030179806,"threshold_uncertainty_score":0.052958548},"labels":[],"label_agreement":null},{"id":"W2616321189","doi":"","title":"An Information System to Prevent Adverse Frug-Food Interactions","year":2009,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Business","score_opus":0.008127253224558947,"score_gpt":0.2819615928587966,"score_spread":0.27383433963423764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2616321189","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24183252,0.0025037965,0.44518435,0.00594632,0.0018197135,0.0034204533,0.012623482,0.24419418,0.042475216],"genre_scores_gemma":[0.5373515,0.0012656314,0.39873466,0.0025339667,0.0006193331,0.0009913428,0.014146716,0.0018107647,0.04254613],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99928087,0.0001727592,0.00009461143,0.00013815795,0.00025950483,0.0000542348],"domain_scores_gemma":[0.995236,0.002108853,0.00058337214,0.0004941029,0.0012719048,0.0003057166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001555619,0.0010773441,0.0006220153,0.0026637397,0.00097431947,0.0012309849,0.0011124494,0.0013357436,0.00935798],"category_scores_gemma":[0.005955666,0.00029174113,0.00040277178,0.0008658131,0.00027298412,0.0018453144,0.0007601163,0.0007002775,0.0049772537],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026595357,0.0018277406,0.01587071,0.00076315616,0.00013202873,0.0011111824,0.00046014314,0.0040574335,0.04459699,0.0033366403,0.14812446,0.7770599],"study_design_scores_gemma":[0.0014756165,0.0041469866,0.062098958,0.00066519773,0.0017608564,0.003022408,0.0010516695,0.33498082,0.24946868,0.012827912,0.32798004,0.0005207959],"about_ca_topic_score_codex":0.003071793,"about_ca_topic_score_gemma":0.0034975123,"teacher_disagreement_score":0.00935798,"about_ca_system_score_codex":0.00057947554,"about_ca_system_score_gemma":0.0016612515,"threshold_uncertainty_score":0.03130561},"labels":[],"label_agreement":null},{"id":"W2620680811","doi":"10.5220/0006294904210431","title":"A LRAAM-based Partial Order Function for Ontology Matching in the Context of Service Discovery","year":2017,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"HEC Montréal","funders":"","keywords":"Ontology; Computer science; Context (archaeology); Matching (statistics); Information retrieval; Function (biology); Order (exchange); Service (business); Service discovery; World Wide Web; Web service; Mathematics; Epistemology; History; Business; Statistics","score_opus":0.027764242982738947,"score_gpt":0.31544474870273176,"score_spread":0.2876805057199928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2620680811","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011329282,0.00029811185,0.98223567,0.000244888,0.00008471292,0.00015544995,0.0006736833,0.0034274904,0.0015507282],"genre_scores_gemma":[0.14084552,0.00020770605,0.8538267,0.00015344273,0.000072512055,0.00019104946,0.001723192,0.0003185224,0.0026613427],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99607205,0.0010947433,0.00055141747,0.0006508799,0.0013355496,0.00029534625],"domain_scores_gemma":[0.9959369,0.0015787986,0.00026163226,0.001122811,0.0009245085,0.00017531619],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003518386,0.00058061694,0.001648301,0.0037898622,0.0014500859,0.003028408,0.0021666887,0.0014209956,0.0046423594],"category_scores_gemma":[0.011073507,0.00042334828,0.002218749,0.0032930733,0.0007543672,0.0048329197,0.0022694631,0.0016850653,0.002164639],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005904729,0.0004930437,0.0034311141,0.00061352056,0.00017570276,0.00037631593,0.0005063428,0.051907834,0.017695483,0.12651163,0.01156521,0.7861333],"study_design_scores_gemma":[0.00004131122,0.00018180994,0.0011100547,0.000087727414,0.00012399141,0.00044539085,0.00020926255,0.85932404,0.017346693,0.10485232,0.016189476,0.00008795864],"about_ca_topic_score_codex":0.00879194,"about_ca_topic_score_gemma":0.01110548,"teacher_disagreement_score":0.00879194,"about_ca_system_score_codex":0.0019407765,"about_ca_system_score_gemma":0.004540994,"threshold_uncertainty_score":0.018607259},"labels":[],"label_agreement":null},{"id":"W2621499279","doi":"10.1037/xlm0000455","title":"Exploring the self-ownership effect: Separating stimulus and response biases.","year":2017,"lang":"en","type":"article","venue":"Journal of Experimental Psychology Learning Memory and Cognition","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Economic and Social Research Council","keywords":"Categorization; Cognitive psychology; Stimulus (psychology); PsycINFO; Psychology; Perception; Prioritization; Object (grammar); Information processing; Task (project management); Social psychology; Computer science; Artificial intelligence; Business; Economics; Political science; MEDLINE","score_opus":0.09530499901207543,"score_gpt":0.3885107134879918,"score_spread":0.29320571447591637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2621499279","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94606,0.00048317958,0.047472123,0.0002898183,0.000026139876,0.0002156612,0.00024472317,0.00011922841,0.005089012],"genre_scores_gemma":[0.985678,0.0001295329,0.013182986,0.00009248351,0.00001414141,0.00009411128,0.00012294481,0.000038233225,0.00064750796],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986046,0.0005294449,0.00009976773,0.0003381657,0.00036044317,0.00006763174],"domain_scores_gemma":[0.967036,0.025402045,0.0038325053,0.0025766075,0.00063730543,0.00051563437],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052538626,0.0002691492,0.00044850027,0.00062244764,0.00021221432,0.0010264267,0.00056828087,0.00065435254,0.003348615],"category_scores_gemma":[0.037593555,0.00026212557,0.0004264566,0.0004290213,0.0012789906,0.003061712,0.0016570315,0.00076482166,0.00027334248],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0057922783,0.0010472705,0.18098299,0.0017642301,0.00042431627,0.000365339,0.0041095857,0.0031977994,0.4446036,0.02164857,0.0008071836,0.3352569],"study_design_scores_gemma":[0.0002993298,0.0033922233,0.66791564,0.00017914483,0.0004395271,0.0013120736,0.0011913339,0.10915091,0.15572712,0.057278726,0.0029888556,0.00012512004],"about_ca_topic_score_codex":0.00082378916,"about_ca_topic_score_gemma":0.0012778577,"teacher_disagreement_score":0.0052538626,"about_ca_system_score_codex":0.00057165936,"about_ca_system_score_gemma":0.00049485744,"threshold_uncertainty_score":0.02778542},"labels":[],"label_agreement":null},{"id":"W2696068122","doi":"10.18653/v1/w17-2407","title":"Extract with Order for Coherent Multi-Document Summarization","year":2017,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"Natural Sciences and Engineering Research Council of Canada; University of Lethbridge","keywords":"Automatic summarization; Readability; Computer science; Coherence (philosophical gambling strategy); Rank (graph theory); Natural language processing; Key (lock); Artificial intelligence; Information retrieval; Selection (genetic algorithm); Sentence; Multi-document summarization; Mathematics","score_opus":0.02839285591499323,"score_gpt":0.3345133148343654,"score_spread":0.3061204589193722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2696068122","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015465946,0.0018786163,0.9733107,0.00032762613,0.00023884814,0.0001608427,0.0011904245,0.0062040263,0.001222987],"genre_scores_gemma":[0.18563451,0.0012718755,0.7950914,0.00031159242,0.0006311071,0.0003149605,0.00783077,0.0008190071,0.008094859],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987739,0.00029781042,0.00014783599,0.00030631394,0.00038238376,0.000091658396],"domain_scores_gemma":[0.9970414,0.0011119802,0.00035594057,0.00057921075,0.0008016568,0.00010968291],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014338379,0.0014351454,0.0010952924,0.002749308,0.00055103976,0.0017477684,0.0008088999,0.0008387951,0.0037579192],"category_scores_gemma":[0.0055039,0.00043784312,0.00090194977,0.001721069,0.00035667163,0.0025432804,0.0010836079,0.001494873,0.0037903355],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053179526,0.0002071846,0.0010949419,0.000725102,0.0001872671,0.00027337883,0.00046661543,0.021277191,0.061154943,0.009721301,0.023688732,0.88067156],"study_design_scores_gemma":[0.00020784272,0.0014604126,0.0039112708,0.00016661496,0.00060482015,0.00072326633,0.0006016661,0.7499978,0.12280843,0.052941483,0.06641773,0.0001587533],"about_ca_topic_score_codex":0.0009832521,"about_ca_topic_score_gemma":0.002777559,"teacher_disagreement_score":0.0037579192,"about_ca_system_score_codex":0.00046316462,"about_ca_system_score_gemma":0.000981883,"threshold_uncertainty_score":0.012571514},"labels":[],"label_agreement":null},{"id":"W2716951971","doi":"10.1109/ccece.2017.7946724","title":"Keyword and Keyphrase Extraction using Newton's Law of Universal Gravitation","year":2017,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Automatic summarization; Word (group theory); Keyword extraction; Newton's law of universal gravitation; Character (mathematics); Weighting; Artificial intelligence; Information retrieval; Text generation; Natural language processing; Word lists by frequency; Task (project management); Gravitation; Sentence; Linguistics","score_opus":0.02751827136155425,"score_gpt":0.3366664469700549,"score_spread":0.3091481756085006,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2716951971","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02112361,0.0010825738,0.96827805,0.00024849473,0.00009355489,0.00026409654,0.0007225257,0.006503346,0.0016837544],"genre_scores_gemma":[0.19304073,0.0011682843,0.7976873,0.0001357727,0.00017990734,0.00027680254,0.0023255832,0.00044270567,0.004742955],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986755,0.00019282453,0.00013375291,0.0003787814,0.00054025475,0.000078865625],"domain_scores_gemma":[0.9981376,0.00064652134,0.00030287515,0.0002665169,0.00058395474,0.0000625082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086306036,0.001377213,0.0010619613,0.005187926,0.00074851536,0.001509497,0.0009968717,0.000988287,0.001864974],"category_scores_gemma":[0.004866611,0.00043629194,0.0013587711,0.00336001,0.0007237552,0.0026030925,0.0008393991,0.000849015,0.0036851575],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030663863,0.00010002587,0.0028206934,0.00056166475,0.0001400313,0.00059307384,0.00063684356,0.019612225,0.06913654,0.011351444,0.00957405,0.8851667],"study_design_scores_gemma":[0.000075837495,0.00036321307,0.005812878,0.00007698917,0.00014371157,0.0019956876,0.00055453717,0.8363475,0.0729787,0.03702569,0.04447146,0.00015378352],"about_ca_topic_score_codex":0.0045433417,"about_ca_topic_score_gemma":0.004778641,"teacher_disagreement_score":0.005187926,"about_ca_system_score_codex":0.00088813825,"about_ca_system_score_gemma":0.0013785629,"threshold_uncertainty_score":0.009033799},"labels":[],"label_agreement":null},{"id":"W272747895","doi":"10.1353/ils.2014.0016","title":"Information Behaviour Research: Where Have We Been, Where Are We Going? / La recherche en comportement informationnel : D’où nous venons, vers quoi nous nous dirigeons?","year":2014,"lang":"fr","type":"article","venue":"Canadian Journal of Information and Library Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Nous; Field (mathematics); Humanities; Computer science; Mathematics; Philosophy","score_opus":0.09874961997334779,"score_gpt":0.32889974048400283,"score_spread":0.23015012051065503,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W272747895","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45463276,0.13036695,0.023762625,0.31133887,0.0017197096,0.00046620806,0.0025592828,0.00033168844,0.07482188],"genre_scores_gemma":[0.9096697,0.051826395,0.017059306,0.010543539,0.0009408338,0.00043863512,0.001052527,0.00025150002,0.008217537],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9612346,0.024191927,0.002502641,0.0019445444,0.008227082,0.001899182],"domain_scores_gemma":[0.8575056,0.084024064,0.013668709,0.0077663017,0.030007511,0.007027746],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.049613696,0.00037005113,0.0008933031,0.0076788636,0.0041420246,0.015488139,0.0013566557,0.0030394804,0.004587586],"category_scores_gemma":[0.094479255,0.00056989264,0.00083757145,0.014869271,0.00836387,0.023528473,0.0026319781,0.0033381125,0.0019619882],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033183282,0.00047311906,0.20996831,0.007577999,0.00027665912,0.000339088,0.23281622,0.00046181565,0.0047134897,0.08333829,0.031261876,0.4284413],"study_design_scores_gemma":[0.000028688415,0.0004973283,0.31120086,0.014546776,0.00016141676,0.00097446336,0.3184344,0.001111823,0.0047952514,0.03552461,0.31241694,0.0003075572],"about_ca_topic_score_codex":0.030313311,"about_ca_topic_score_gemma":0.026288128,"teacher_disagreement_score":0.9503863,"about_ca_system_score_codex":0.011476028,"about_ca_system_score_gemma":0.014776103,"threshold_uncertainty_score":0.26238543},"labels":[],"label_agreement":null},{"id":"W2736357342","doi":"10.3934/bdia.2017001","title":"First steps in the investigation of automated text annotation with pictures","year":2017,"lang":"en","type":"article","venue":"Big Data and Information Analytics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Annotation; Computer science; Natural language processing; Artificial intelligence; Information retrieval; Computer graphics (images)","score_opus":0.0545208875358242,"score_gpt":0.2972021565263602,"score_spread":0.24268126899053596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2736357342","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027734999,0.0011644154,0.94813186,0.003058414,0.00020818524,0.002829319,0.00066252745,0.002824201,0.013386085],"genre_scores_gemma":[0.04873119,0.00039703865,0.94509345,0.0003779969,0.00006004079,0.0011578536,0.00058539474,0.00040097014,0.0031960646],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98866874,0.0060464293,0.0005989442,0.0017251929,0.0024624767,0.000498133],"domain_scores_gemma":[0.97180474,0.014753112,0.000858583,0.0052470393,0.006878838,0.00045771882],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010170407,0.0020348423,0.0012327436,0.004097397,0.0034593246,0.00828172,0.004243892,0.004189819,0.010240392],"category_scores_gemma":[0.03607284,0.0018546802,0.0016327337,0.003281824,0.004661132,0.014387183,0.004433576,0.0049050706,0.0042863293],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011162776,0.0018169776,0.0070400834,0.0040845093,0.00022780107,0.0026155612,0.014441453,0.00993445,0.17747085,0.17090698,0.017382804,0.5929623],"study_design_scores_gemma":[0.00027065616,0.0018980369,0.010781414,0.001351751,0.00023024045,0.0042143515,0.014080438,0.16909021,0.3675706,0.15215147,0.27775195,0.00060892705],"about_ca_topic_score_codex":0.006467499,"about_ca_topic_score_gemma":0.005394255,"teacher_disagreement_score":0.010240392,"about_ca_system_score_codex":0.0027572603,"about_ca_system_score_gemma":0.0036085013,"threshold_uncertainty_score":0.053786874},"labels":[],"label_agreement":null},{"id":"W2740737347","doi":"10.5539/ibr.v10n9p1","title":"An Integrated Methodology for Approaching Sentiment Analysis in Business Domain","year":2017,"lang":"en","type":"article","venue":"International Business Research","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Sentiment analysis; Computer science; Domain (mathematical analysis); Artificial intelligence; Natural language processing; Data science; Data mining; Mathematics","score_opus":0.19898083347813106,"score_gpt":0.5018635052710343,"score_spread":0.30288267179290324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740737347","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005787269,0.0000646199,0.9968528,0.00010231809,0.00004802486,0.00023055648,0.00012864357,0.00072514795,0.0012691187],"genre_scores_gemma":[0.008003555,0.00012635883,0.9894488,0.00006587723,0.00004098009,0.00037694725,0.0003630155,0.00013879206,0.001435707],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99616903,0.0012497026,0.0005559706,0.00061414484,0.0012683067,0.00014282543],"domain_scores_gemma":[0.9971891,0.0008310847,0.00028207904,0.0004083128,0.0011575148,0.00013193357],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050996323,0.0011507148,0.000683448,0.0045358646,0.0010750121,0.0040798755,0.0010775335,0.0009805855,0.004838634],"category_scores_gemma":[0.006539491,0.00068627147,0.0018719112,0.0028760731,0.0008543688,0.0030638697,0.0024340705,0.0019297245,0.004333331],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000094693256,0.00025694945,0.0022317313,0.0010271427,0.0002516878,0.0006334359,0.002101551,0.007697311,0.03846407,0.16953844,0.017417435,0.7602856],"study_design_scores_gemma":[0.00009374311,0.00029619766,0.0040275366,0.0007681716,0.00025659768,0.0017933619,0.0014795299,0.24507113,0.038562696,0.35108584,0.3563358,0.00022945159],"about_ca_topic_score_codex":0.0009972696,"about_ca_topic_score_gemma":0.001337401,"teacher_disagreement_score":0.0050996323,"about_ca_system_score_codex":0.0007653484,"about_ca_system_score_gemma":0.0020756214,"threshold_uncertainty_score":0.02696979},"labels":[],"label_agreement":null},{"id":"W2746131889","doi":"10.1016/j.jpain.2004.02.573","title":"Other","year":2004,"lang":"en","type":"article","venue":"Journal of Pain","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Chronic pain; Cognition; Salient; Multidimensional scaling; Medicine; Psychology; Cognitive psychology; Clinical psychology; Applied psychology; Physical therapy; Psychiatry; Artificial intelligence; Computer science; Machine learning","score_opus":0.009906544577876932,"score_gpt":0.27415197258420504,"score_spread":0.2642454280063281,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2746131889","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005851964,0.00030744978,0.003928458,0.0013102572,0.001438151,0.00017563142,0.008044291,0.005857193,0.9783533],"genre_scores_gemma":[0.0036332214,0.0003746917,0.002243291,0.0011554302,0.00042986157,0.000117614036,0.009655357,0.0018619025,0.98052853],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99914503,0.000092211056,0.000039112147,0.00014627114,0.00046159336,0.000115858515],"domain_scores_gemma":[0.9965702,0.00038064786,0.00014151396,0.0009830067,0.0011741144,0.0007505741],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0010020025,0.0009888889,0.00060615427,0.003178222,0.001483036,0.004027917,0.0018671838,0.0012683639,0.8195795],"category_scores_gemma":[0.0051609157,0.0003403552,0.00072787184,0.001775275,0.00038586342,0.0024857551,0.0031228352,0.001152364,0.758615],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006330811,0.000054253356,0.00029611966,0.000112618174,0.000006180392,0.000062744904,0.000060682763,0.000056366272,0.0008065373,0.0049457713,0.869428,0.12410755],"study_design_scores_gemma":[0.000012488416,0.0000092136515,0.00034256856,0.00003811199,0.0000050002304,0.00007088725,0.000031875115,0.0000917574,0.0004148208,0.0015815416,0.9973947,0.000007049337],"about_ca_topic_score_codex":0.0022676543,"about_ca_topic_score_gemma":0.004429911,"teacher_disagreement_score":0.18042052,"about_ca_system_score_codex":0.0007779503,"about_ca_system_score_gemma":0.0016191332,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2752756221","doi":"10.3233/aac-170027","title":"Ontological representations of rhetorical figures for argument mining","year":2017,"lang":"en","type":"article","venue":"Argument & Computation","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Science and Engineering Research Board; University of Waterloo","keywords":"Rhetorical question; Argument (complex analysis); Computer science; Linguistics; Epistemology; Natural language processing; Philosophy; Medicine","score_opus":0.06141192465762198,"score_gpt":0.3821481057053004,"score_spread":0.3207361810476784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2752756221","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0065728347,0.000660023,0.9748107,0.0016198464,0.00012970757,0.00032641212,0.0020700938,0.001033121,0.012777254],"genre_scores_gemma":[0.08982019,0.0007798676,0.901221,0.00022503646,0.00008435149,0.0005510258,0.0047222436,0.0002486666,0.0023476426],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9964561,0.0015339917,0.0005220279,0.0003971,0.00093211106,0.00015865582],"domain_scores_gemma":[0.9940644,0.0032543405,0.00053900795,0.0013355444,0.0006072998,0.00019936601],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004009389,0.0009543184,0.00081779447,0.009890767,0.0028650872,0.007227654,0.0023935155,0.0017175849,0.0072152787],"category_scores_gemma":[0.018948872,0.00081515196,0.0030263138,0.009433295,0.0024423886,0.011962343,0.0035021105,0.003443321,0.0024990207],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000032532957,0.000095955846,0.0014778164,0.0003883463,0.00006018787,0.00023008163,0.002639896,0.008685062,0.001242041,0.8828461,0.0061208503,0.096181124],"study_design_scores_gemma":[0.000018528619,0.000021661303,0.0008985723,0.00047741295,0.00007328329,0.0002649147,0.0020918509,0.0977225,0.0019225848,0.7633438,0.1331115,0.000053415955],"about_ca_topic_score_codex":0.004859195,"about_ca_topic_score_gemma":0.007971955,"teacher_disagreement_score":0.009890767,"about_ca_system_score_codex":0.0031193513,"about_ca_system_score_gemma":0.0025325527,"threshold_uncertainty_score":0.024137557},"labels":[],"label_agreement":null},{"id":"W2765761355","doi":"10.1007/978-3-319-67837-5_11","title":"Opinions Sandbox: Turning Emotions on Topics into Actionable Analytics","year":2017,"lang":"en","type":"book-chapter","venue":"Lecture notes of the Institute for Computer Sciences, Social Informatics and Telecommunications Engineering","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Sandbox (software development); Computer science; Sentiment analysis; Analytics; sort; Product (mathematics); Service (business); Plan (archaeology); Data science; World Wide Web; Operations research; Artificial intelligence; Information retrieval; Engineering; Marketing; Business; Software engineering","score_opus":0.028024601738707862,"score_gpt":0.2834433201999242,"score_spread":0.25541871846121633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2765761355","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028196665,0.0031093082,0.82488245,0.003920018,0.0025127626,0.00037153292,0.005455624,0.04086364,0.09068796],"genre_scores_gemma":[0.29463,0.002734227,0.5619224,0.0022035623,0.0012261473,0.0008802387,0.01063309,0.0112259295,0.114544556],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994118,0.00020260767,0.00003135702,0.000142239,0.0001730452,0.00003903721],"domain_scores_gemma":[0.9985209,0.0009461489,0.0000668117,0.00013953188,0.00019021613,0.00013641836],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009331519,0.0012496576,0.00051391806,0.001036184,0.0006154651,0.0031416186,0.0009114124,0.0008082303,0.032258496],"category_scores_gemma":[0.0056765494,0.00036105685,0.000803227,0.00085072534,0.0006465282,0.0049260035,0.0024786403,0.0016674807,0.012898847],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078993774,0.00020266528,0.0016648176,0.0008520944,0.00012448168,0.00026171567,0.0031814564,0.0039701965,0.029165883,0.074119724,0.22227544,0.66339165],"study_design_scores_gemma":[0.00013917431,0.00021719272,0.0041761915,0.0004573192,0.00021789355,0.00024536683,0.002390769,0.15798241,0.03562326,0.3658335,0.43255752,0.00015946313],"about_ca_topic_score_codex":0.0013038601,"about_ca_topic_score_gemma":0.0025734066,"teacher_disagreement_score":0.032258496,"about_ca_system_score_codex":0.00055650354,"about_ca_system_score_gemma":0.00046941906,"threshold_uncertainty_score":0.10791534},"labels":[],"label_agreement":null},{"id":"W2767742613","doi":"10.5406/amerjpsyc.130.4.0401","title":"S. S. Stevens’s Invariant Legacy: Scale Types and the Power Law","year":2017,"lang":"en","type":"article","venue":"The American Journal of Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Psychology; Sketch; Perception; Power (physics); Psychophysics; Epistemology; Cognitive psychology; Cognitive science; Computer science; Philosophy; Algorithm","score_opus":0.013219934637145946,"score_gpt":0.3316426457040686,"score_spread":0.31842271106692266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2767742613","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03810922,0.11209857,0.23264948,0.17769091,0.027388604,0.0001058395,0.00073457754,0.0005649642,0.41065785],"genre_scores_gemma":[0.7725511,0.042723093,0.057929024,0.023515204,0.041693233,0.00017704876,0.00034231643,0.0006740292,0.06039491],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.997717,0.0006416364,0.00015942162,0.00060758874,0.0007830419,0.000091309215],"domain_scores_gemma":[0.9872803,0.008433926,0.00069277367,0.0011819154,0.002047928,0.00036313495],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036445165,0.00060342083,0.00078242936,0.0031868902,0.0018193395,0.004650095,0.0006810449,0.001544944,0.006058231],"category_scores_gemma":[0.023535155,0.00046739733,0.000782954,0.002285809,0.010146236,0.007867858,0.0017247694,0.004687212,0.0016458741],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001724274,0.000009604008,0.00040599296,0.000052486055,0.000011224261,0.000044263157,0.00030813133,0.0003344029,0.00016870105,0.9584196,0.020290947,0.019937519],"study_design_scores_gemma":[0.0000065619993,0.000018507848,0.0010611372,0.00009953257,0.000011325128,0.00014843837,0.00010007042,0.0011386463,0.00021428395,0.92774665,0.069425024,0.000029869167],"about_ca_topic_score_codex":0.0021839277,"about_ca_topic_score_gemma":0.001140369,"teacher_disagreement_score":0.006058231,"about_ca_system_score_codex":0.0022412702,"about_ca_system_score_gemma":0.0008514301,"threshold_uncertainty_score":0.02026683},"labels":[],"label_agreement":null},{"id":"W2770119650","doi":"10.1002/asi.23980","title":"Improving interpretations of topic modeling in microblogs","year":2017,"lang":"en","type":"article","venue":"Journal of the Association for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University; McGill University; Ontario Tech University; Concordia University","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; King Saud University; Saudi Arabian Cultural Bureau; CRC Health Group","keywords":"Computer science; Perplexity; Topic model; Information retrieval; Microblogging; WordNet; Social media; Process (computing); Latent Dirichlet allocation; Coherence (philosophical gambling strategy); Natural language processing; Artificial intelligence; Data science; World Wide Web; Language model","score_opus":0.010995341567729244,"score_gpt":0.2826981526934637,"score_spread":0.27170281112573447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2770119650","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07631307,0.0022876994,0.91519904,0.0010996215,0.0001889278,0.00013402815,0.0006038317,0.0016453397,0.0025284276],"genre_scores_gemma":[0.7254445,0.0017920827,0.26605877,0.00029311565,0.0007419508,0.00030572305,0.0023758712,0.00063000864,0.0023580384],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99455476,0.003372809,0.00032913504,0.0008834446,0.0005985686,0.00026124076],"domain_scores_gemma":[0.96705806,0.027609667,0.0011521607,0.0019161932,0.001901002,0.00036294857],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009799724,0.002044625,0.0016674926,0.005562925,0.0015416251,0.0051196986,0.0018742149,0.001881694,0.0016226419],"category_scores_gemma":[0.04798796,0.0012209609,0.0018576044,0.003592884,0.0013663612,0.008371036,0.002919619,0.0030422849,0.0009919925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018252549,0.00043824813,0.03151545,0.0010905187,0.00069768104,0.0008939439,0.010357938,0.3435694,0.0109258145,0.07929463,0.013922219,0.50546896],"study_design_scores_gemma":[0.000033986827,0.000037019618,0.0017092708,0.00005295634,0.000057957903,0.00006473723,0.00051213615,0.9431258,0.0016110016,0.04966113,0.0030946096,0.000039402254],"about_ca_topic_score_codex":0.008116923,"about_ca_topic_score_gemma":0.0092546465,"teacher_disagreement_score":0.009799724,"about_ca_system_score_codex":0.0019182384,"about_ca_system_score_gemma":0.0013648155,"threshold_uncertainty_score":0.051826477},"labels":[],"label_agreement":null},{"id":"W2783735091","doi":"10.1109/bigdata.2017.8258529","title":"Big data in psychology: Using word embeddings to study theory-of-mind","year":2017,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Concreteness; Word (group theory); Reliability (semiconductor); Reading (process); Big data; Computer science; Cognitive psychology; Natural language processing; Psychology; Artificial intelligence; Cognitive science; Linguistics; Data mining; Philosophy","score_opus":0.19015671293255137,"score_gpt":0.4556427411724217,"score_spread":0.26548602823987033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2783735091","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12270767,0.0032469435,0.86416715,0.0035728419,0.0004780293,0.00024572603,0.0016178032,0.0005158895,0.0034479562],"genre_scores_gemma":[0.6741192,0.0010061222,0.32180834,0.00044991617,0.00025103928,0.00055329606,0.0013070164,0.00011575521,0.00038947005],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99510854,0.0033041905,0.00030787138,0.0005514034,0.0006369753,0.00009094482],"domain_scores_gemma":[0.91167957,0.073629886,0.00410354,0.0071697347,0.0025368894,0.0008803328],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006819345,0.00090330344,0.0008502821,0.005594592,0.0009425123,0.0034742283,0.0010435173,0.0013006303,0.0011643426],"category_scores_gemma":[0.075061746,0.00045006498,0.0012005388,0.005240828,0.0027103429,0.008493736,0.0034531117,0.0026591127,0.00023378718],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004900971,0.0007128417,0.101773895,0.0021535708,0.0016341418,0.00041043645,0.01167158,0.043989427,0.005580198,0.38069814,0.0070588454,0.44382685],"study_design_scores_gemma":[0.000034475826,0.00019038316,0.012857447,0.00029618008,0.00012873214,0.00018609517,0.002103996,0.1737736,0.002005633,0.801578,0.006757269,0.00008819038],"about_ca_topic_score_codex":0.0011403587,"about_ca_topic_score_gemma":0.0012303084,"teacher_disagreement_score":0.006819345,"about_ca_system_score_codex":0.0010270557,"about_ca_system_score_gemma":0.0009351831,"threshold_uncertainty_score":0.036064625},"labels":[],"label_agreement":null},{"id":"W2784269545","doi":"10.1007/s10115-017-1147-9","title":"Localized user-driven topic discovery via boosted ensemble of nonnegative matrix factorization","year":2018,"lang":"en","type":"article","venue":"Knowledge and Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Research Foundation of Korea","keywords":"Matrix decomposition; Computer science; Factorization; Matrix (chemical analysis); Non-negative matrix factorization; Mathematics; Artificial intelligence; Algorithm; Physics","score_opus":0.009477174681492609,"score_gpt":0.27222000813632263,"score_spread":0.26274283345483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2784269545","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026083862,0.0013050697,0.9687497,0.0004799293,0.0003150174,0.00009856149,0.00047086593,0.0014632125,0.0010338527],"genre_scores_gemma":[0.5172196,0.0013075038,0.46705967,0.000573029,0.0014297295,0.00039342412,0.004744101,0.00041598317,0.0068571116],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972031,0.001004944,0.00014272452,0.0007511267,0.00058541447,0.00031266618],"domain_scores_gemma":[0.99375665,0.0035376637,0.00033690146,0.0006434082,0.0013773915,0.00034794872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032289773,0.00179771,0.0025947369,0.0025829573,0.0013075193,0.0018941684,0.0026271609,0.0020872282,0.002144259],"category_scores_gemma":[0.009759893,0.00075553556,0.0020301696,0.0026097416,0.0007990813,0.0037012007,0.002373175,0.002362843,0.002109441],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002054002,0.0012876759,0.007850456,0.00070858997,0.0007308743,0.0004143031,0.0009896205,0.12702367,0.04567176,0.017101744,0.03632962,0.7598377],"study_design_scores_gemma":[0.000025516436,0.000060221606,0.00046105238,0.000013762197,0.000064433356,0.000073199124,0.000045821787,0.98703045,0.002513256,0.008176457,0.0015110135,0.000024798217],"about_ca_topic_score_codex":0.005197224,"about_ca_topic_score_gemma":0.010697315,"teacher_disagreement_score":0.005197224,"about_ca_system_score_codex":0.00058626255,"about_ca_system_score_gemma":0.0018017602,"threshold_uncertainty_score":0.017076671},"labels":[],"label_agreement":null},{"id":"W2786269463","doi":"10.3390/jrfm11010008","title":"Estimation of Cross-Lingual News Similarities Using Text-Mining Methods","year":2018,"lang":"en","type":"article","venue":"Journal of risk and financial management","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Similarity (geometry); Natural language processing; Artificial intelligence; Word (group theory); Information retrieval; Task (project management); The Internet; World Wide Web; Image (mathematics); Linguistics","score_opus":0.0254576792545957,"score_gpt":0.3778261132054529,"score_spread":0.3523684339508572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2786269463","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21594045,0.0015331405,0.7749548,0.00028683335,0.00019790106,0.00032262382,0.0016988683,0.0014326241,0.0036327974],"genre_scores_gemma":[0.63897955,0.00054757507,0.3524143,0.00011857986,0.0003409465,0.0003757328,0.0048693656,0.00015673363,0.0021971771],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961281,0.00086356746,0.0006630408,0.0013691587,0.00079695205,0.00017910049],"domain_scores_gemma":[0.99159944,0.0040961755,0.0012669275,0.00072067004,0.0021499482,0.00016685568],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028869212,0.0010611675,0.0011166588,0.009321133,0.0008162822,0.0017344995,0.0014266024,0.0011084498,0.0012938601],"category_scores_gemma":[0.0124216555,0.00045238633,0.0013217485,0.0062810327,0.00043159016,0.003933813,0.0013055893,0.0009191993,0.0013155469],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068541046,0.00074288587,0.057372235,0.00068874966,0.00082334015,0.001021243,0.00095277926,0.03738548,0.019036349,0.0056342143,0.0048154267,0.8708419],"study_design_scores_gemma":[0.00006304165,0.0002295637,0.026646974,0.000061087834,0.00027666488,0.0008294902,0.0008338565,0.9383366,0.01673703,0.009145886,0.006747639,0.000092183516],"about_ca_topic_score_codex":0.0019028516,"about_ca_topic_score_gemma":0.0027027503,"teacher_disagreement_score":0.009321133,"about_ca_system_score_codex":0.00055086514,"about_ca_system_score_gemma":0.00074314,"threshold_uncertainty_score":0.01526767},"labels":[],"label_agreement":null},{"id":"W2786731957","doi":"10.18653/v1/w17-5532","title":"Generating and Evaluating Summaries for Partial Email Threads: Conversational Bayesian Surprise and Silver Standards","year":2017,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Automatic summarization; Surprise; Thread (computing); Annotation; Redundancy (engineering); Information retrieval; Bayesian probability; Natural language processing; Artificial intelligence; Machine learning; Programming language","score_opus":0.03591123173986498,"score_gpt":0.3603061999791383,"score_spread":0.32439496823927333,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2786731957","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.063563645,0.0019865893,0.9178174,0.001037443,0.00028806308,0.00032185708,0.0029587084,0.0092846,0.0027417326],"genre_scores_gemma":[0.31922552,0.00048166225,0.6610881,0.00039896017,0.00048864144,0.0005166376,0.012293056,0.0011051493,0.0044022063],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9943879,0.00232409,0.0005502873,0.001439271,0.0010380074,0.0002603973],"domain_scores_gemma":[0.9753288,0.0119115645,0.00211251,0.0045233145,0.0052198167,0.0009040222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00724354,0.0014102048,0.0016637287,0.0037470344,0.0014743845,0.003109115,0.0019593802,0.0023824887,0.0028052027],"category_scores_gemma":[0.047678396,0.0005856736,0.0010608146,0.0024546934,0.00076990353,0.0053834086,0.0028694964,0.0021639047,0.0024006926],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001483437,0.00033011404,0.011976983,0.00093298353,0.00027436664,0.0003859538,0.0030494053,0.029811954,0.018413156,0.01958476,0.036305267,0.87745154],"study_design_scores_gemma":[0.00016388038,0.00062849757,0.0071034357,0.00021320368,0.00024971718,0.0005017476,0.0012041058,0.8349401,0.029356685,0.08005232,0.04543042,0.00015586152],"about_ca_topic_score_codex":0.003068378,"about_ca_topic_score_gemma":0.0063544116,"teacher_disagreement_score":0.00724354,"about_ca_system_score_codex":0.0013800373,"about_ca_system_score_gemma":0.0021122925,"threshold_uncertainty_score":0.038307965},"labels":[],"label_agreement":null},{"id":"W2792698203","doi":"10.3115/v1/w14-21","title":"Proceedings of the First Workshop on Argumentation Mining","year":2014,"lang":"en","type":"paratext","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Joint Information Systems Committee; European Commission; National Science Foundation","keywords":"Argumentation theory; Computer science; Data science; Epistemology; Philosophy","score_opus":0.016015825415811577,"score_gpt":0.28411471498866686,"score_spread":0.26809888957285527,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2792698203","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0110614505,0.040584046,0.72424847,0.05234921,0.021557884,0.0012332816,0.009818143,0.008257154,0.13089037],"genre_scores_gemma":[0.07077381,0.028888674,0.6082367,0.0052054264,0.008694202,0.001798955,0.04960409,0.0049819723,0.22181618],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9915354,0.0037474306,0.0009782519,0.0016846432,0.0017148588,0.00033937822],"domain_scores_gemma":[0.980613,0.011293449,0.00038832324,0.0037223522,0.0028157716,0.0011671054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012524304,0.0018171628,0.0021684284,0.0045855157,0.0021512143,0.013374175,0.0042788326,0.0031682388,0.071831204],"category_scores_gemma":[0.029406996,0.0010784417,0.00404921,0.005605099,0.0022238025,0.014394095,0.005732217,0.005918387,0.026124079],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028295664,0.0003258506,0.00084243715,0.0010336916,0.00016974202,0.0003448035,0.0010908779,0.0032798168,0.0013814762,0.0589815,0.40304372,0.52922314],"study_design_scores_gemma":[0.000059444297,0.00005396111,0.0009747014,0.00069457234,0.000057067828,0.00036710495,0.0005356162,0.013187451,0.0014734166,0.095737785,0.88681215,0.00004661562],"about_ca_topic_score_codex":0.0026197308,"about_ca_topic_score_gemma":0.0031847125,"teacher_disagreement_score":0.071831204,"about_ca_system_score_codex":0.0028519363,"about_ca_system_score_gemma":0.0035044977,"threshold_uncertainty_score":0.24029928},"labels":[],"label_agreement":null},{"id":"W2793407730","doi":"10.5220/0006664405520559","title":"Modeling a Tool for Conducting Systematic Reviews Iteratively","year":2018,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Systems engineering; Engineering","score_opus":0.17384002596433243,"score_gpt":0.3854992197691817,"score_spread":0.2116591938048493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2793407730","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017117078,0.00096385507,0.9839225,0.0011104203,0.00013169399,0.0035074283,0.0024858182,0.0046471483,0.0015194423],"genre_scores_gemma":[0.009769379,0.00028326735,0.98405486,0.00009048797,0.000025981024,0.0046782484,0.00069202966,0.00012154698,0.00028422964],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.87321144,0.09718663,0.014920176,0.0060542193,0.008071096,0.0005564909],"domain_scores_gemma":[0.41696745,0.53720975,0.014400483,0.014632095,0.015488457,0.001301654],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.13065541,0.00392097,0.004315354,0.015001827,0.002513519,0.0063520367,0.004725567,0.0027969915,0.014015297],"category_scores_gemma":[0.4160222,0.0039691,0.009886103,0.010930299,0.0011953834,0.006779386,0.0048873695,0.0038414942,0.002909751],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014729939,0.00035240987,0.0068849134,0.0402608,0.009452061,0.0006319768,0.006689564,0.08279278,0.0029387185,0.13875161,0.035054386,0.6747178],"study_design_scores_gemma":[0.0014573688,0.00092856336,0.0017401583,0.01237152,0.009331574,0.00061479,0.0010396952,0.522754,0.00728696,0.33867142,0.10337085,0.0004331144],"about_ca_topic_score_codex":0.006709521,"about_ca_topic_score_gemma":0.017034048,"teacher_disagreement_score":0.8693446,"about_ca_system_score_codex":0.004588544,"about_ca_system_score_gemma":0.023485426,"threshold_uncertainty_score":0.6909801},"labels":[],"label_agreement":null},{"id":"W2802218424","doi":"10.14288/1.0365983","title":"Multimodal human brain connectivity analysis based on graph theory","year":2018,"lang":"en","type":"article","venue":"cIRcle (University of British Columbia)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Graph theory; Artificial intelligence; Mathematics; Combinatorics","score_opus":0.007363979779461453,"score_gpt":0.21290133771178843,"score_spread":0.20553735793232697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2802218424","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021701751,0.00016210413,0.97627676,0.0001200852,0.000008637391,0.00004382515,0.00013640973,0.00023197703,0.0013183883],"genre_scores_gemma":[0.67161524,0.000734875,0.32372046,0.00010235909,0.00006907328,0.0002470933,0.00068653136,0.00018552715,0.0026388776],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99968624,0.000120467674,0.000010437342,0.00007993812,0.00007585153,0.00002706399],"domain_scores_gemma":[0.99948055,0.00028547697,0.00007705912,0.000051043495,0.00007698287,0.000028935227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048220076,0.00057428237,0.0004901932,0.0021014863,0.00032946162,0.0006283506,0.0005719752,0.00046885738,0.0023332126],"category_scores_gemma":[0.0023520086,0.00022813573,0.00080237404,0.0010170087,0.00065435166,0.0011310094,0.0006943061,0.0005361683,0.0002664431],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000060391132,0.000044326847,0.0018883115,0.00015913373,0.00012089849,0.00020207872,0.00017015405,0.7864976,0.017049076,0.089645855,0.0022969486,0.10186519],"study_design_scores_gemma":[0.0000020219313,0.000015641413,0.0008939679,0.0000045779925,0.0000084976355,0.000032539745,0.000011698544,0.9745076,0.0005999934,0.023383772,0.00053263997,0.0000071871323],"about_ca_topic_score_codex":0.004953915,"about_ca_topic_score_gemma":0.0049869358,"teacher_disagreement_score":0.004953915,"about_ca_system_score_codex":0.0007140382,"about_ca_system_score_gemma":0.00046091864,"threshold_uncertainty_score":0.009850144},"labels":[],"label_agreement":null},{"id":"W2806157614","doi":"10.1075/term.00012.amj","title":"Distributed specificity for automatic terminology extraction","year":2018,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; Carleton University; University of Ottawa","funders":"","keywords":"Computer science; Terminology; Artificial intelligence; Classifier (UML); Filter (signal processing); Natural language processing; Representation (politics); Domain (mathematical analysis); Pattern recognition (psychology); Computer vision; Linguistics; Mathematics","score_opus":0.017873356744103102,"score_gpt":0.3525913017325484,"score_spread":0.33471794498844526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806157614","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024848908,0.00044479698,0.96895057,0.00025435706,0.00006028469,0.00010526065,0.0003088927,0.0024031848,0.0026236894],"genre_scores_gemma":[0.36361834,0.00033759652,0.6291032,0.00020829882,0.00016268839,0.00024861787,0.0022923376,0.0005003171,0.0035285412],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99662256,0.0013266595,0.00028727518,0.00085207954,0.0006975078,0.00021386056],"domain_scores_gemma":[0.9931479,0.0029756182,0.000680967,0.0016213789,0.0013958458,0.0001782438],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023762702,0.0009913505,0.0010784294,0.0054465076,0.000901314,0.0028650262,0.0012108393,0.0010742869,0.0033119284],"category_scores_gemma":[0.011815781,0.00045369944,0.0010880304,0.004297035,0.0010365394,0.0044529918,0.0033527156,0.0017508083,0.0030261483],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023955292,0.00010744656,0.0038349635,0.00040034857,0.0001291936,0.00022463346,0.0006224069,0.0095972065,0.065653965,0.034882512,0.0076452917,0.8766625],"study_design_scores_gemma":[0.00009367132,0.0002695493,0.007945208,0.00020725957,0.0002182325,0.0013603522,0.0009424115,0.6510552,0.08586602,0.20602132,0.045893263,0.00012750918],"about_ca_topic_score_codex":0.00083295803,"about_ca_topic_score_gemma":0.0011658958,"teacher_disagreement_score":0.0054465076,"about_ca_system_score_codex":0.00087062496,"about_ca_system_score_gemma":0.0013256972,"threshold_uncertainty_score":0.012567103},"labels":[],"label_agreement":null},{"id":"W2810540096","doi":"10.3233/aac-180037","title":"An annotation scheme for Rhetorical Figures","year":2018,"lang":"en","type":"article","venue":"Argument & Computation","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Social Sciences and Humanities Research Council of Canada; University of Waterloo","keywords":"Rhetorical question; Annotation; Scheme (mathematics); Computer science; Natural language processing; Linguistics; Artificial intelligence; Mathematics; Philosophy","score_opus":0.028278597947934452,"score_gpt":0.35779315040429455,"score_spread":0.3295145524563601,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2810540096","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009198661,0.00028216734,0.93852264,0.0032815465,0.0011085417,0.00033658408,0.0024847048,0.005158837,0.03962638],"genre_scores_gemma":[0.10056335,0.00040093856,0.8720481,0.000790039,0.00042134346,0.000591502,0.003418221,0.0016386749,0.020127881],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.993646,0.0025595932,0.00094144477,0.0011398705,0.0013942523,0.0003188363],"domain_scores_gemma":[0.9750004,0.009535926,0.0015566417,0.0075466875,0.005646658,0.0007136894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007821249,0.00075441215,0.00065733783,0.006890336,0.0042899125,0.0057481453,0.0022926203,0.003276324,0.019951414],"category_scores_gemma":[0.027780486,0.0010352742,0.0009868358,0.0046217004,0.0041836184,0.01281909,0.0052919784,0.0039091767,0.0101401275],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012617672,0.000034924717,0.0011127449,0.00022206052,0.000009919353,0.0002675388,0.0037245145,0.0010801229,0.0046506794,0.86278754,0.019464638,0.10651908],"study_design_scores_gemma":[0.000035225647,0.000079120866,0.0007788497,0.00042197574,0.000044992932,0.0006916987,0.0009509429,0.015553295,0.009109879,0.3436363,0.62859553,0.00010220426],"about_ca_topic_score_codex":0.0014788412,"about_ca_topic_score_gemma":0.0012701307,"teacher_disagreement_score":0.019951414,"about_ca_system_score_codex":0.002213794,"about_ca_system_score_gemma":0.0022078583,"threshold_uncertainty_score":0.06674409},"labels":[],"label_agreement":null},{"id":"W2810801822","doi":"10.3390/s18072117","title":"Using Stigmergy to Distinguish Event-Specific Topics in Social Discussions","year":2018,"lang":"en","type":"article","venue":"Sensors","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Stigmergy; Event (particle physics); Computer science; Data science; Psychology; Human–computer interaction; Cognitive science; Communication; Artificial intelligence; Physics","score_opus":0.05025072984723933,"score_gpt":0.3580322272384279,"score_spread":0.3077814973911886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2810801822","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6526486,0.0008079453,0.34051532,0.00024453603,0.00008846973,0.00024560504,0.00034650875,0.00080231635,0.004300669],"genre_scores_gemma":[0.9434175,0.00015826184,0.055271804,0.00002735256,0.00002940463,0.00007084708,0.00023898443,0.000054074004,0.0007317727],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99929094,0.0001801635,0.00006439126,0.00023424476,0.00016308922,0.00006708242],"domain_scores_gemma":[0.9965714,0.0019264717,0.00067398365,0.00034328055,0.00027337697,0.00021147088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011847888,0.0005565046,0.00057432754,0.0040416038,0.00066226436,0.0019489446,0.00058353535,0.0007763019,0.0012474766],"category_scores_gemma":[0.01019766,0.00034887888,0.0006433976,0.0021559566,0.0010586441,0.0034463594,0.0015072038,0.00062901323,0.00041178468],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023502272,0.0005275286,0.100903496,0.0010379658,0.0005130743,0.0006897528,0.006061507,0.24329878,0.072475746,0.049927846,0.0022520327,0.5199619],"study_design_scores_gemma":[0.000045117944,0.00034366615,0.0633957,0.00006308687,0.00012796975,0.0004498387,0.0011627428,0.8644791,0.02057744,0.044613276,0.00460493,0.00013714623],"about_ca_topic_score_codex":0.0018532454,"about_ca_topic_score_gemma":0.0024242303,"teacher_disagreement_score":0.0040416038,"about_ca_system_score_codex":0.0009190903,"about_ca_system_score_gemma":0.0004972459,"threshold_uncertainty_score":0.006668508},"labels":[],"label_agreement":null},{"id":"W2826552311","doi":"","title":"Real-time Change Point Detection using On-line Topic Models.","year":2018,"lang":"en","type":"article","venue":"NPARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Latent Dirichlet allocation; Change detection; Line (geometry); Computer science; Point (geometry); Bayesian probability; Social media; Topic model; Data mining; Artificial intelligence; Mathematics; World Wide Web","score_opus":0.07014499979451327,"score_gpt":0.31555338110763737,"score_spread":0.2454083813131241,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2826552311","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0766012,0.0013959004,0.9051766,0.00041039335,0.00018359012,0.00033910593,0.0029949779,0.009662667,0.0032354817],"genre_scores_gemma":[0.609892,0.0007242055,0.37543973,0.00021956202,0.000323526,0.0005134064,0.008251546,0.0007648748,0.00387113],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978231,0.0006447392,0.0001285857,0.0006352461,0.00062410993,0.00014415233],"domain_scores_gemma":[0.9904145,0.0053329403,0.0012760523,0.0010492008,0.0016694199,0.00025775947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002481376,0.0011675177,0.00076650566,0.0045300983,0.00052249094,0.0017300426,0.001328603,0.0012067276,0.0019316005],"category_scores_gemma":[0.011744977,0.00043569267,0.0009421612,0.0026617,0.00040279594,0.002303689,0.0013517869,0.0014229076,0.0030725014],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079797005,0.0005942649,0.040610865,0.0005570819,0.0003634141,0.00054939213,0.0014030765,0.03328792,0.04058478,0.0028027121,0.017657245,0.8607913],"study_design_scores_gemma":[0.0000651791,0.0002212918,0.027940413,0.000063781416,0.00014571668,0.00067992765,0.00068458146,0.9063647,0.030272637,0.011111118,0.02236848,0.0000822229],"about_ca_topic_score_codex":0.0035031773,"about_ca_topic_score_gemma":0.0046073697,"teacher_disagreement_score":0.0045300983,"about_ca_system_score_codex":0.00049616385,"about_ca_system_score_gemma":0.000497774,"threshold_uncertainty_score":0.013122916},"labels":[],"label_agreement":null},{"id":"W2883250117","doi":"10.6084/m9.figshare.c.4174325.v1","title":"Supplementary material from \"Appetitive information seeking behaviour reveals robust daily rhythmicity for Internet-based food-related keyword searches\"","year":2018,"lang":"en","type":"article","venue":"Figshare","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"The Internet; Behaviour change; Internet privacy; Computer science; Advertising; Psychology; World Wide Web; Business","score_opus":0.03570539376612959,"score_gpt":0.2759296156537625,"score_spread":0.24022422188763293,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2883250117","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004050853,0.0001977699,0.0021023205,0.000572225,0.0005417193,0.00014662754,0.977409,0.0020107091,0.0129687935],"genre_scores_gemma":[0.014466595,0.0004554257,0.008255536,0.0007681023,0.0001765213,0.00078543933,0.95207226,0.0011869363,0.021833062],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993005,0.00008537953,0.000087340086,0.00015925778,0.00027209794,0.0000954792],"domain_scores_gemma":[0.9944019,0.0027914643,0.00031574906,0.0005458314,0.001615158,0.0003299524],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00081091415,0.00092385506,0.0010097355,0.0022783806,0.0014403797,0.002008615,0.00139484,0.001092409,0.65518105],"category_scores_gemma":[0.011847196,0.000494213,0.00092405843,0.0036039008,0.00024876816,0.001569563,0.0017146181,0.00076634495,0.2063577],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005909018,0.0001569048,0.0039849984,0.0018303837,0.000072438765,0.00015751095,0.00018637687,0.00027224925,0.0025217764,0.0011266159,0.96036285,0.028737113],"study_design_scores_gemma":[0.0006949949,0.00039386476,0.101441234,0.0016801828,0.00016701843,0.0009894273,0.0007105604,0.002956367,0.004171439,0.0093649,0.87720263,0.00022746847],"about_ca_topic_score_codex":0.010954472,"about_ca_topic_score_gemma":0.024491072,"teacher_disagreement_score":0.65518105,"about_ca_system_score_codex":0.00077203405,"about_ca_system_score_gemma":0.0018778596,"threshold_uncertainty_score":0.4918424},"labels":[],"label_agreement":null},{"id":"W2884251704","doi":"10.46867/ijcp.2018.31.00.01","title":"Erratum: How and Why Does Category Learning Cause Categorical Perception?","year":2018,"lang":"en","type":"erratum","venue":"International Journal of Comparative Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; McGill University","funders":"","keywords":"Perception; Categorical variable; Psychology; Cognitive psychology; Communication; Computer science; Neuroscience; Machine learning","score_opus":0.04067674895692886,"score_gpt":0.3931664117920442,"score_spread":0.35248966283511535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2884251704","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012318995,0.0012161861,0.002031769,0.17613076,0.80383515,0.00003405713,0.002326119,0.00054264994,0.012651415],"genre_scores_gemma":[0.06834406,0.007159515,0.013337784,0.1887367,0.23643567,0.00025105992,0.0064209434,0.0026795308,0.47663465],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99693334,0.00037950714,0.0006012081,0.00046177165,0.0014258961,0.00019821317],"domain_scores_gemma":[0.96626717,0.010264272,0.0018106926,0.0020810873,0.018711291,0.0008655331],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003215775,0.0009681899,0.0009362182,0.001729445,0.0031054264,0.002352917,0.001906465,0.0053144256,0.07301767],"category_scores_gemma":[0.06426356,0.0005355444,0.0007380142,0.0012267856,0.002366922,0.002775944,0.0017682449,0.0056712953,0.031629313],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005666071,0.000011621593,0.00017766579,0.00007754045,0.000006817306,0.00030443174,0.0000755893,0.00003266906,0.00010329871,0.002660029,0.9880872,0.008406459],"study_design_scores_gemma":[0.000041725292,0.00004235518,0.0013106228,0.00033663076,0.000044065106,0.0011990337,0.00049392576,0.00046562735,0.0012116078,0.010190941,0.9846073,0.00005616578],"about_ca_topic_score_codex":0.008603237,"about_ca_topic_score_gemma":0.009361138,"teacher_disagreement_score":0.07301767,"about_ca_system_score_codex":0.0026324012,"about_ca_system_score_gemma":0.0031284585,"threshold_uncertainty_score":0.24426842},"labels":[],"label_agreement":null},{"id":"W2889229100","doi":"","title":"NLP for Conversations: Sentiment, Summarization, and Group Dynamics","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of the Fraser Valley","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Sentiment analysis; Artificial intelligence; Dynamics (music); Information retrieval; Group (periodic table); Psychology","score_opus":0.029641051308958492,"score_gpt":0.33400512286884554,"score_spread":0.30436407155988704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2889229100","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08452837,0.0016383958,0.8854877,0.004536615,0.0004917681,0.00044633713,0.0077240746,0.0052576396,0.009889054],"genre_scores_gemma":[0.601488,0.0010192469,0.37319246,0.00032710438,0.0009167192,0.0006941078,0.01658889,0.0007849846,0.0049884454],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99582857,0.0023078432,0.00033182662,0.00069848186,0.0006552272,0.00017807771],"domain_scores_gemma":[0.9825434,0.012961539,0.0009928436,0.0016012174,0.0015892233,0.0003117538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004400794,0.00094884593,0.0008258212,0.0025368144,0.0014940177,0.0025575005,0.0010702127,0.0010789571,0.0051638247],"category_scores_gemma":[0.029656868,0.00044421072,0.0008051083,0.0028670838,0.0005975799,0.006166846,0.0020517635,0.0020275754,0.003252694],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009171231,0.00042487917,0.007998055,0.0013369176,0.00024514107,0.00037642338,0.0062032402,0.016714819,0.026298074,0.03730067,0.045698114,0.8564866],"study_design_scores_gemma":[0.00008897978,0.00024361901,0.009763062,0.00025383904,0.00018429795,0.00029153138,0.0037598235,0.74730897,0.016254889,0.17883307,0.042912673,0.000105294406],"about_ca_topic_score_codex":0.0022860994,"about_ca_topic_score_gemma":0.0023020555,"teacher_disagreement_score":0.0051638247,"about_ca_system_score_codex":0.00090152066,"about_ca_system_score_gemma":0.0010714118,"threshold_uncertainty_score":0.023273885},"labels":[],"label_agreement":null},{"id":"W2890039419","doi":"10.1007/978-3-030-00066-0_11","title":"Venue Classification of Research Papers in Scholarly Digital Libraries","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; Canadian Institute for Advanced Research","keywords":"Computer science; Digital library; Information retrieval; Library science; World Wide Web; Art","score_opus":0.04333766805299147,"score_gpt":0.318591590279443,"score_spread":0.2752539222264515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890039419","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2229062,0.16176203,0.0129185915,0.007339798,0.015375143,0.001327014,0.21510814,0.007714305,0.35554877],"genre_scores_gemma":[0.46337822,0.07435702,0.0332638,0.0019841178,0.00860422,0.0008358747,0.19845271,0.0051546567,0.21396942],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99407417,0.0011522791,0.00085584726,0.000605063,0.00241373,0.00089885504],"domain_scores_gemma":[0.970182,0.008765807,0.004340513,0.0017139618,0.008158623,0.006839068],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0026649397,0.00088295795,0.0012182177,0.079704106,0.002393363,0.0094343815,0.0016963119,0.0011056414,0.059533194],"category_scores_gemma":[0.029723436,0.0003643029,0.0013638189,0.06526236,0.0006703997,0.0042698784,0.0034737843,0.0009958047,0.033164166],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014551866,0.00026069392,0.108089834,0.011230486,0.0005999458,0.00096107705,0.002719684,0.0010292892,0.00973881,0.015094608,0.30675986,0.5420605],"study_design_scores_gemma":[0.00009120227,0.0002509607,0.16798311,0.0040967707,0.00035678322,0.0022469761,0.0063397125,0.0014504441,0.0035424614,0.009017603,0.80449903,0.00012488123],"about_ca_topic_score_codex":0.002858023,"about_ca_topic_score_gemma":0.0071881814,"teacher_disagreement_score":0.9202959,"about_ca_system_score_codex":0.0016759599,"about_ca_system_score_gemma":0.003242089,"threshold_uncertainty_score":0.19915837},"labels":[],"label_agreement":null},{"id":"W2892333563","doi":"10.1108/jarhe-03-2018-0047","title":"Digital library keyword analysis for visualization education research","year":2018,"lang":"en","type":"article","venue":"Journal of Applied Research in Higher Education","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"MacEwan University","funders":"","keywords":"Visualization; Computer science; Digital library; Information retrieval; Relevance (law); Information visualization; Thesaurus; World Wide Web; Selection (genetic algorithm); Data science; Data mining; Artificial intelligence","score_opus":0.11472020053815307,"score_gpt":0.4795430512056294,"score_spread":0.36482285066747633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2892333563","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19243556,0.25161648,0.22561312,0.026152888,0.004496195,0.05427862,0.04310949,0.0052058147,0.1970918],"genre_scores_gemma":[0.32314408,0.07677129,0.5076176,0.0059891585,0.0010004579,0.056909744,0.014431318,0.0012836122,0.012852817],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.92269534,0.038807612,0.02097284,0.0028016581,0.013686059,0.0010364157],"domain_scores_gemma":[0.5730878,0.31186503,0.034071997,0.016815519,0.061551426,0.0026082864],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.06778465,0.0011209383,0.0032134769,0.052628107,0.0027912639,0.016352173,0.0026042862,0.0013974175,0.021723878],"category_scores_gemma":[0.31797567,0.0008270753,0.0032042284,0.04965438,0.0020356742,0.013072391,0.0062269964,0.0016219938,0.0069808513],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009644268,0.00036167886,0.020010488,0.14120105,0.0009157163,0.000571953,0.0130533185,0.0009263381,0.0034936974,0.02747653,0.032341525,0.7586833],"study_design_scores_gemma":[0.000809218,0.002764517,0.051078446,0.226626,0.005901586,0.0031467902,0.04961962,0.0063776798,0.01911152,0.056425665,0.57747865,0.00066028314],"about_ca_topic_score_codex":0.0046229656,"about_ca_topic_score_gemma":0.0071681994,"teacher_disagreement_score":0.9473719,"about_ca_system_score_codex":0.0082257595,"about_ca_system_score_gemma":0.02318643,"threshold_uncertainty_score":0.3584838},"labels":[],"label_agreement":null},{"id":"W2894138074","doi":"10.1167/18.10.971","title":"The Impact of Self-Relevance and Valence on Word Processing: an ERP study","year":2018,"lang":"en","type":"article","venue":"Journal of Vision","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Psychology; Valence (chemistry); Trait; Cognition; Cognitive psychology; Recall; Emotional valence; Event-related potential; Cognitive bias; Developmental psychology; Neuroscience","score_opus":0.014434976634239921,"score_gpt":0.3768346529064032,"score_spread":0.3623996762721633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2894138074","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9995012,0.00003408384,0.00020426759,0.0000048569264,0.0000021461453,0.000007888188,0.000017279048,0.0000026135094,0.00022563941],"genre_scores_gemma":[0.99876213,0.000079600766,0.00065599056,0.000027745913,0.000011004128,0.000023526776,0.00007065437,0.000009145435,0.00036025996],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9999138,0.000017294587,0.0000060870416,0.000026323458,0.000026584545,0.000009908325],"domain_scores_gemma":[0.99946994,0.0002786998,0.00009310771,0.00004779027,0.00005236155,0.000058214544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00023970354,0.00017217676,0.00019081918,0.0001433748,0.00008810274,0.00026206163,0.0001079229,0.00018175348,0.001192714],"category_scores_gemma":[0.0012290804,0.000107682055,0.00009750551,0.00012220757,0.00023857968,0.00022568622,0.00018736959,0.0002555494,0.00016336747],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033114704,0.0012526223,0.0211045,0.00016023254,0.000037662154,0.0003521213,0.0009781801,0.00007172038,0.9500303,0.00020451743,0.000098887795,0.022397798],"study_design_scores_gemma":[0.00026977167,0.0060848016,0.9174294,0.000013399221,0.00014260397,0.0015670058,0.0005701483,0.0012462736,0.070914164,0.0008013686,0.0009363241,0.000024933448],"about_ca_topic_score_codex":0.00016383952,"about_ca_topic_score_gemma":0.00023242585,"teacher_disagreement_score":0.001192714,"about_ca_system_score_codex":0.000060831484,"about_ca_system_score_gemma":0.00008233888,"threshold_uncertainty_score":0.003990054},"labels":[],"label_agreement":null},{"id":"W2895058937","doi":"","title":"Multiple In-text Reference Phenomenon","year":2016,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Phenomenon; Computer science; Epistemology; Philosophy","score_opus":0.019466237835809606,"score_gpt":0.25161854705288045,"score_spread":0.23215230921707083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2895058937","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36388266,0.007367197,0.2578543,0.015835568,0.0043273084,0.0004748416,0.004487139,0.01010479,0.3356662],"genre_scores_gemma":[0.929632,0.0012712167,0.017252056,0.0012317665,0.0026361367,0.00013642671,0.0021814061,0.001872358,0.043786693],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9916113,0.0026180807,0.00059155026,0.0019046343,0.002860512,0.00041389852],"domain_scores_gemma":[0.9218937,0.041272018,0.00557968,0.017994685,0.011595358,0.0016645835],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.004673535,0.0007273122,0.0011732236,0.0066166213,0.0031755522,0.006562692,0.0021600917,0.0029908833,0.051889498],"category_scores_gemma":[0.055809457,0.00052101415,0.0005949589,0.008001279,0.002305916,0.013650717,0.0040076924,0.0026769803,0.010511876],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014898777,0.0003230597,0.013298247,0.0031044448,0.00026947484,0.005845442,0.037713684,0.0019361305,0.07051782,0.42796937,0.10224673,0.33528566],"study_design_scores_gemma":[0.00029244428,0.00054766046,0.040576395,0.0009679494,0.0006176085,0.011845881,0.015307811,0.04231396,0.07803466,0.275934,0.5331967,0.00036487065],"about_ca_topic_score_codex":0.001063283,"about_ca_topic_score_gemma":0.00074456056,"teacher_disagreement_score":0.9933834,"about_ca_system_score_codex":0.001265111,"about_ca_system_score_gemma":0.0009740777,"threshold_uncertainty_score":0.17358768},"labels":[],"label_agreement":null},{"id":"W2895115809","doi":"10.1145/3209280.3229100","title":"Automatic Term Extraction in Technical Domain using Part-of-Speech and Common-Word Features","year":2018,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Terminology; Term (time); Domain (mathematical analysis); Confusion; Technical documentation; Documentation; Key (lock); Natural language processing; Ambiguity; Artificial intelligence; Information retrieval; Word (group theory); Speech recognition; Programming language; Linguistics","score_opus":0.02079483283180173,"score_gpt":0.3401692492987932,"score_spread":0.3193744164669915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2895115809","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052527893,0.0022750122,0.90853775,0.00049293437,0.0003421801,0.00067829236,0.009692616,0.020629274,0.004824043],"genre_scores_gemma":[0.10167413,0.00097248884,0.8698932,0.00014537683,0.00014078773,0.00050942454,0.021583373,0.0011288918,0.0039523267],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99784136,0.0004135069,0.0004477506,0.0005385902,0.0006409317,0.000117893294],"domain_scores_gemma":[0.99247897,0.003948049,0.0006863445,0.000721311,0.0020288306,0.00013652194],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014833749,0.0020473143,0.0011752241,0.010876468,0.0010799443,0.0016725718,0.0011608757,0.001544416,0.0041979817],"category_scores_gemma":[0.008540659,0.00049754395,0.0013987261,0.005721747,0.00056102296,0.0025060808,0.0013813279,0.0012802777,0.005070153],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029548063,0.00015747249,0.0028517477,0.0020596038,0.00011138477,0.0013456149,0.0008406907,0.0040428475,0.20600949,0.0048827385,0.022734523,0.7546684],"study_design_scores_gemma":[0.00021068842,0.00054086355,0.028428998,0.0005867878,0.0005693694,0.007418093,0.0018971093,0.33170855,0.42870894,0.02295839,0.17657381,0.0003983427],"about_ca_topic_score_codex":0.0029691688,"about_ca_topic_score_gemma":0.0037795876,"teacher_disagreement_score":0.010876468,"about_ca_system_score_codex":0.00066249724,"about_ca_system_score_gemma":0.0020295253,"threshold_uncertainty_score":0.014043629},"labels":[],"label_agreement":null},{"id":"W2895553377","doi":"10.21105/joss.00774","title":"quanteda: An R package for the quantitative analysis of textual data","year":2018,"lang":"en","type":"article","venue":"The Journal of Open Source Software","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1323,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"De Beers (Canada)","funders":"London School of Economics and Political Science","keywords":"R package; Computer science; Natural language processing; Information retrieval; Programming language","score_opus":0.12209439307154113,"score_gpt":0.4211929597768534,"score_spread":0.29909856670531226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2895553377","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005810041,0.0014394451,0.47543687,0.0011593217,0.0006601073,0.0007521333,0.3189017,0.18856902,0.007271363],"genre_scores_gemma":[0.04798994,0.0013187807,0.614308,0.0014026604,0.00041911157,0.006595136,0.17116769,0.14553921,0.011259466],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99465275,0.0022979234,0.0005183062,0.0011326518,0.0012164345,0.00018199187],"domain_scores_gemma":[0.9680011,0.023643175,0.002284335,0.0032854846,0.002339555,0.00044626498],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067070327,0.0024419425,0.002382296,0.004722014,0.00086692756,0.0040590637,0.0029489498,0.0008442505,0.07443252],"category_scores_gemma":[0.06362126,0.0014601982,0.0025948675,0.004280706,0.0010171375,0.0028200184,0.0030585465,0.0028480783,0.04489454],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076938124,0.000100196805,0.011011511,0.007867895,0.003391676,0.0005506496,0.00089806155,0.008467674,0.006437853,0.018841058,0.8058477,0.1358163],"study_design_scores_gemma":[0.0004562131,0.00019918769,0.018046573,0.00092965097,0.0012256036,0.0010504595,0.0003217137,0.048994564,0.009679314,0.08912782,0.82959795,0.00037098033],"about_ca_topic_score_codex":0.0029308186,"about_ca_topic_score_gemma":0.004392474,"teacher_disagreement_score":0.07443252,"about_ca_system_score_codex":0.00073824805,"about_ca_system_score_gemma":0.0031061072,"threshold_uncertainty_score":0.24900156},"labels":[],"label_agreement":null},{"id":"W2901340518","doi":"10.3389/fpsyg.2018.02185","title":"Concise, Simple, and Not Wrong: In Search of a Short-Hand Interpretation of Statistical Significance","year":2018,"lang":"en","type":"article","venue":"Frontiers in Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Interpretation (philosophy); Simple (philosophy); Meaning (existential); Psychology; Relevance (law); Cognitive psychology; Computer science; Epistemology","score_opus":0.02158321688355779,"score_gpt":0.3669483879343335,"score_spread":0.3453651710507757,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2901340518","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0060099815,0.0061197393,0.83419496,0.109800495,0.02545435,0.00080298854,0.0015005465,0.0045549055,0.011561982],"genre_scores_gemma":[0.11704102,0.0052822586,0.7895066,0.053351276,0.01899858,0.0033145905,0.001284342,0.004754851,0.0064664953],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.8507276,0.11371502,0.013294119,0.004743084,0.016577039,0.0009431663],"domain_scores_gemma":[0.42190596,0.4636317,0.0323953,0.02810572,0.04990775,0.004053605],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.10819158,0.0038031226,0.0022548693,0.00571952,0.0031250862,0.01331519,0.0044666454,0.00822532,0.0083145285],"category_scores_gemma":[0.51832104,0.0017025847,0.0015137792,0.0028386468,0.015377719,0.0205924,0.0077835596,0.018236384,0.008927296],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001097757,0.00019360418,0.0017078837,0.0070812707,0.00036970025,0.0016280616,0.045876343,0.0020535516,0.011424274,0.4251347,0.33870584,0.16472702],"study_design_scores_gemma":[0.00019461669,0.00035320403,0.0012976209,0.0053905644,0.00019100987,0.0017429645,0.0074449256,0.005944186,0.005306108,0.5684627,0.40338805,0.0002840113],"about_ca_topic_score_codex":0.0006108202,"about_ca_topic_score_gemma":0.0006651008,"teacher_disagreement_score":0.8918084,"about_ca_system_score_codex":0.002520182,"about_ca_system_score_gemma":0.00583706,"threshold_uncertainty_score":0.5721786},"labels":[],"label_agreement":null},{"id":"W2903706655","doi":"","title":"Personalization of an Environmental Message: Developing a Measure","year":2018,"lang":"en","type":"article","venue":"Journal of the Association for Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Personalization; Measure (data warehouse); Computer science; Internet privacy; World Wide Web; Data mining","score_opus":0.010902697356835066,"score_gpt":0.25053837038483,"score_spread":0.23963567302799493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903706655","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6948542,0.0049026194,0.13502182,0.00739167,0.0008850091,0.002730584,0.0019916126,0.0011369217,0.15108553],"genre_scores_gemma":[0.9498674,0.0009798926,0.043659072,0.0005364637,0.0004516626,0.0005744686,0.0003771151,0.000039007948,0.003514883],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9942601,0.0026951332,0.00056248246,0.0005347809,0.001755631,0.00019194547],"domain_scores_gemma":[0.9587423,0.022424089,0.0062902058,0.0026906738,0.008403077,0.0014495293],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007924772,0.0006683121,0.00042804764,0.0028455455,0.0009328884,0.0028943317,0.00080955063,0.0013909659,0.0034546116],"category_scores_gemma":[0.041364465,0.00020373348,0.000500776,0.0017552611,0.0012735849,0.0058771526,0.0023548566,0.0012969275,0.0008755908],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010121802,0.0017472693,0.3549868,0.0016105084,0.00054571946,0.00015660383,0.012229973,0.0040700654,0.009336797,0.023016715,0.009109584,0.58217776],"study_design_scores_gemma":[0.00014283703,0.004820118,0.8141172,0.0012321774,0.00129265,0.0012385652,0.024248127,0.03611981,0.018543694,0.02719809,0.07064381,0.00040298913],"about_ca_topic_score_codex":0.0011840475,"about_ca_topic_score_gemma":0.0012746826,"teacher_disagreement_score":0.007924772,"about_ca_system_score_codex":0.0012282507,"about_ca_system_score_gemma":0.00079324364,"threshold_uncertainty_score":0.04191065},"labels":[],"label_agreement":null},{"id":"W2907738398","doi":"10.1016/j.eswa.2018.12.054","title":"Ranking résumés automatically using only résumés: A method free of job offers","year":2018,"lang":"fr","type":"article","venue":"Expert Systems with Applications","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Consejo Nacional de Ciencia y Tecnología; Association Nationale de la Recherche et de la Technologie","keywords":"Ranking (information retrieval); Computer science; Relevance (law); Rank (graph theory); Information retrieval; Similarity (geometry); Job analysis; Vocabulary; Selection (genetic algorithm); Process (computing); Learning to rank; Resource (disambiguation); Artificial intelligence; Machine learning; Mathematics; Linguistics; Job satisfaction","score_opus":0.05402005072200111,"score_gpt":0.37502874975065953,"score_spread":0.3210086990286584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2907738398","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022790216,0.0014271816,0.7724781,0.00044267252,0.0007018218,0.00072924217,0.012933599,0.1767385,0.011758742],"genre_scores_gemma":[0.15446484,0.0005147258,0.75245017,0.00022769072,0.00065393664,0.0005541154,0.019865945,0.010601874,0.060666725],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981336,0.00026652604,0.00017130503,0.00050370314,0.0007815406,0.0001433166],"domain_scores_gemma":[0.9941683,0.0022877916,0.00035400604,0.0015379536,0.001290834,0.00036115418],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014843378,0.0018839148,0.0018459071,0.005625961,0.0011120156,0.0028250283,0.00220683,0.0013537935,0.028238574],"category_scores_gemma":[0.009490781,0.0008216814,0.0012005292,0.0036530907,0.0004001645,0.0036344167,0.0020121185,0.0014223028,0.024492107],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001204078,0.00039513156,0.002801002,0.0005730352,0.0001572934,0.00022323197,0.00019431148,0.0019416523,0.021305298,0.0028456568,0.07307483,0.8952845],"study_design_scores_gemma":[0.0008424144,0.00094937754,0.019797863,0.00030716546,0.00072797615,0.002264124,0.0008524201,0.52820194,0.10079199,0.03699915,0.30773896,0.00052661315],"about_ca_topic_score_codex":0.00213444,"about_ca_topic_score_gemma":0.0055632233,"teacher_disagreement_score":0.028238574,"about_ca_system_score_codex":0.00032973627,"about_ca_system_score_gemma":0.0018156198,"threshold_uncertainty_score":0.0944674},"labels":[],"label_agreement":null},{"id":"W2908347907","doi":"10.4000/books.aaccademia.4620","title":"UNIBA - Integrating distributional semantics features in a supervised approach for detecting irony in Italian tweets","year":2018,"lang":"en","type":"book-chapter","venue":"Accademia University Press eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Research Council Canada; Università degli Studi di Napoli Federico II","keywords":"Task (project management); Irony; Computer science; Artificial intelligence; Constraint (computer-aided design); Natural language processing; Representation (politics); Semantics (computer science); Polarity (international relations); Distributional semantics; Machine learning; Information retrieval; Semantic similarity; Linguistics; Mathematics; Engineering","score_opus":0.025360972556899394,"score_gpt":0.24859290899076839,"score_spread":0.223231936433869,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2908347907","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22937754,0.0030738586,0.5130941,0.00260092,0.00095452776,0.00087600324,0.024004104,0.17376721,0.05225167],"genre_scores_gemma":[0.5992439,0.0005502576,0.31578267,0.0006267806,0.00048496018,0.00070558593,0.04480634,0.0031151958,0.034684364],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99907875,0.0003286554,0.000047727393,0.00025210952,0.00019589216,0.00009675969],"domain_scores_gemma":[0.9988574,0.00044353007,0.00007960652,0.00029229783,0.0002447206,0.00008240969],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013001442,0.0014509445,0.0006483176,0.0024894467,0.0010286199,0.0015697044,0.0009405319,0.0007503573,0.0060693584],"category_scores_gemma":[0.0027958385,0.00051029853,0.00079072185,0.0012214797,0.00048130157,0.0025655278,0.0016780962,0.0013579761,0.007994981],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007136383,0.0009029737,0.016552934,0.00067970424,0.00025598094,0.00041162974,0.0007627754,0.009224365,0.038042657,0.0056880554,0.12870228,0.79806304],"study_design_scores_gemma":[0.00019049451,0.00054819894,0.034355484,0.00016290565,0.00019804836,0.0011033544,0.0010291811,0.7610954,0.05001501,0.03778171,0.113332,0.00018825488],"about_ca_topic_score_codex":0.004705283,"about_ca_topic_score_gemma":0.016310409,"teacher_disagreement_score":0.0060693584,"about_ca_system_score_codex":0.0007588761,"about_ca_system_score_gemma":0.0012055242,"threshold_uncertainty_score":0.020304024},"labels":[],"label_agreement":null},{"id":"W2908681319","doi":"10.17821/srels/2018/v55i6/132490","title":"The Accuracy of Newspaper Citation Count Reported and Actual Citation Found in Web of Science Citation Database","year":2018,"lang":"en","type":"article","venue":"SRELS Journal of Information Management","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Science North","funders":"","keywords":"Citation; Newspaper; Scopus; Citation database; Web of science; Computer science; Information retrieval; Citation analysis; Database; World Wide Web; Library science; Political science; MEDLINE","score_opus":0.0229233944683186,"score_gpt":0.315035029832838,"score_spread":0.2921116353645194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2908681319","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90714926,0.013884776,0.019012414,0.0017150732,0.0009704758,0.00015473906,0.024701178,0.00071973697,0.03169234],"genre_scores_gemma":[0.97244745,0.0028890322,0.010008934,0.00013889073,0.00042297377,0.00007844333,0.01112535,0.00010744636,0.0027815166],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9672368,0.007493242,0.008806236,0.0030146863,0.012706559,0.00074247184],"domain_scores_gemma":[0.7687134,0.14058778,0.02571542,0.018496137,0.04567634,0.00081103697],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.016915262,0.00034419712,0.0008233897,0.019491065,0.00075170206,0.0037752087,0.0013429705,0.0009182868,0.0026058883],"category_scores_gemma":[0.17298037,0.00022164387,0.00058460754,0.023644606,0.000640198,0.002792262,0.00094996486,0.00054653874,0.0018622059],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044820763,0.00019472158,0.77195185,0.0020932555,0.000511232,0.00037296856,0.0026535185,0.0025409672,0.0035552727,0.0024930139,0.0059230616,0.207262],"study_design_scores_gemma":[0.000031648713,0.0004145505,0.89837897,0.00084035355,0.0008277094,0.001580408,0.003935219,0.020109877,0.020362757,0.0036061264,0.049749494,0.00016292876],"about_ca_topic_score_codex":0.0055435915,"about_ca_topic_score_gemma":0.0042980267,"teacher_disagreement_score":0.98308474,"about_ca_system_score_codex":0.0010387765,"about_ca_system_score_gemma":0.00094947167,"threshold_uncertainty_score":0.08945751},"labels":[],"label_agreement":null},{"id":"W2928032745","doi":"10.1007/s00500-019-03963-y","title":"Automatic keyphrase extraction using word embeddings","year":2019,"lang":"en","type":"article","venue":"Soft Computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"Department of Industrial and Systems Engineering, Hong Kong Polytechnic University; National Natural Science Foundation of China","keywords":"Word (group theory); Computer science; Natural language processing; Artificial intelligence; Extraction (chemistry); Speech recognition; Linguistics","score_opus":0.014090135774292753,"score_gpt":0.3117116156751691,"score_spread":0.29762147990087634,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2928032745","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.070351675,0.0028794466,0.8878647,0.0006782975,0.0012218417,0.00056862656,0.009656134,0.019476766,0.007302539],"genre_scores_gemma":[0.24203382,0.0019443458,0.7238625,0.00016197342,0.00036236344,0.00025971862,0.016818352,0.001435816,0.013121194],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991041,0.00009044401,0.00015076109,0.00023641065,0.00031699112,0.00010126275],"domain_scores_gemma":[0.9975942,0.0006472064,0.0002683955,0.0003403012,0.001050919,0.000099017896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00045576447,0.0015045016,0.0010029953,0.0056461045,0.00063176046,0.0019786535,0.0005908444,0.00083658437,0.008320482],"category_scores_gemma":[0.0036739768,0.00042890848,0.0010084257,0.0043666465,0.00036165366,0.0032937278,0.0013836129,0.0011723171,0.0119255185],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045229975,0.00013570525,0.0016529367,0.0009440388,0.00006621357,0.0005639264,0.00020904855,0.001289238,0.13898024,0.0072237295,0.020658921,0.82782376],"study_design_scores_gemma":[0.00021488035,0.00073499023,0.010853976,0.00041278137,0.000493736,0.004520153,0.0017424166,0.25829887,0.47399184,0.055951305,0.19252361,0.00026147158],"about_ca_topic_score_codex":0.0007437039,"about_ca_topic_score_gemma":0.0011827935,"teacher_disagreement_score":0.008320482,"about_ca_system_score_codex":0.0003342808,"about_ca_system_score_gemma":0.0010263846,"threshold_uncertainty_score":0.027834773},"labels":[],"label_agreement":null},{"id":"W2943217089","doi":"10.1145/3297280.3297382","title":"Study of linguistic features incorporated in a literary book recommender system","year":2019,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Recommender system; Computer science; Pleasure; Reading (process); Key (lock); Natural language processing; Information retrieval; Quality (philosophy); Artificial intelligence; World Wide Web; Linguistics; Psychology","score_opus":0.01113746719009497,"score_gpt":0.2691508151635969,"score_spread":0.25801334797350195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2943217089","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9838483,0.00032307778,0.013046337,0.00034462535,0.000023673681,0.000087868524,0.00017444571,0.00013026637,0.0020213318],"genre_scores_gemma":[0.98946756,0.00006156704,0.009166389,0.000035741792,0.000016038255,0.000025921523,0.00015130687,0.000014641261,0.0010609119],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99900717,0.00053197733,0.000063191685,0.00011746194,0.00022579991,0.000054393546],"domain_scores_gemma":[0.97612107,0.020193003,0.000830616,0.0005093781,0.0021042349,0.00024172959],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019795455,0.00029494785,0.0003915008,0.0010589169,0.0006607468,0.001055388,0.0005392239,0.0005514617,0.0014293853],"category_scores_gemma":[0.019139223,0.00020861182,0.00038561577,0.00078681146,0.00046326462,0.0012868157,0.0002493397,0.0006622618,0.00025420415],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006012977,0.0036432499,0.1821726,0.0027318632,0.00078943564,0.0030583485,0.006275719,0.20894895,0.1774635,0.008218795,0.0049212947,0.39576316],"study_design_scores_gemma":[0.00006881568,0.0015075596,0.06208459,0.000046456298,0.00019698967,0.00038671537,0.001128506,0.9196797,0.012009614,0.0012076773,0.0016193261,0.00006410926],"about_ca_topic_score_codex":0.0073646754,"about_ca_topic_score_gemma":0.008559083,"teacher_disagreement_score":0.0073646754,"about_ca_system_score_codex":0.0007808748,"about_ca_system_score_gemma":0.00036867958,"threshold_uncertainty_score":0.014643669},"labels":[],"label_agreement":null},{"id":"W2945545832","doi":"10.48550/arxiv.1905.07689","title":"DivGraphPointer: A Graph Pointer Network for Extracting Diverse Keyphrases","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Computer science; Automatic summarization; Graph; Information retrieval; Artificial intelligence; Text graph; Pointer (user interface); Natural language processing; Theoretical computer science","score_opus":0.07042154819932844,"score_gpt":0.2193973463951663,"score_spread":0.14897579819583787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945545832","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032550324,0.0014283853,0.92101246,0.00026837553,0.00019297063,0.0003938219,0.005861548,0.0332736,0.0050186063],"genre_scores_gemma":[0.10231365,0.00095107453,0.8626893,0.00022447713,0.00008679599,0.00032297175,0.015450957,0.0014782262,0.016482547],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99968684,0.000034712382,0.000017733071,0.00012638669,0.00010123604,0.00003314927],"domain_scores_gemma":[0.99942636,0.0001801654,0.00007483244,0.00012596094,0.00015754219,0.000035271492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000366725,0.0019723424,0.0008020916,0.004587442,0.0006140083,0.0010332844,0.0013664152,0.0011726068,0.0060866657],"category_scores_gemma":[0.0018668638,0.0004762736,0.00073557213,0.003301994,0.000497668,0.0026821322,0.001241992,0.0012145351,0.004678748],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002564984,0.00012050397,0.0010802174,0.00038563088,0.000098503224,0.00025010802,0.00017671783,0.014722071,0.046184625,0.00598388,0.022547418,0.90819377],"study_design_scores_gemma":[0.00011734876,0.0003680233,0.003290354,0.00010119183,0.00015522305,0.00080029597,0.00035212436,0.76249194,0.10084644,0.039512,0.09182865,0.00013639503],"about_ca_topic_score_codex":0.00489589,"about_ca_topic_score_gemma":0.0139849605,"teacher_disagreement_score":0.0060866657,"about_ca_system_score_codex":0.0006384638,"about_ca_system_score_gemma":0.00093069155,"threshold_uncertainty_score":0.02036196},"labels":[],"label_agreement":null},{"id":"W2946532448","doi":"10.1145/3292500.3330727","title":"A User-Centered Concept Mining System for Query and Document Understanding at Tencent","year":2019,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Knowledge base; Web query classification; Web search query; Web mining; Taxonomy (biology); Query language; Core (optical fiber); Database query","score_opus":0.02961210578107067,"score_gpt":0.2768240722171036,"score_spread":0.2472119664360329,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946532448","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007829893,0.00036825036,0.8701853,0.0006544785,0.00008843304,0.0006707911,0.0045614755,0.11222816,0.0034132088],"genre_scores_gemma":[0.043838594,0.00023937112,0.93591624,0.00071172667,0.00007751569,0.00094712625,0.009777539,0.0016884858,0.0068033393],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99797267,0.0004067425,0.00027699178,0.0006471569,0.0006255199,0.00007092089],"domain_scores_gemma":[0.9949484,0.0022424571,0.0002924268,0.00080773473,0.0013729584,0.00033589674],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044992147,0.0013322034,0.0016087275,0.005481374,0.0014403557,0.0032282176,0.0029512756,0.0019590857,0.012840614],"category_scores_gemma":[0.011195131,0.00071902026,0.0012185798,0.003965943,0.00057939626,0.0066776457,0.0035966157,0.0021225885,0.008959408],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001846649,0.0011663154,0.005762697,0.0009522816,0.00033852618,0.001111669,0.002764553,0.0032539344,0.049686156,0.035778303,0.14219229,0.75514674],"study_design_scores_gemma":[0.00037630703,0.000569651,0.0057607815,0.00028480418,0.00034031444,0.0019205115,0.0010049823,0.5669404,0.06916377,0.08129703,0.27195618,0.0003852015],"about_ca_topic_score_codex":0.0039192964,"about_ca_topic_score_gemma":0.006841011,"teacher_disagreement_score":0.012840614,"about_ca_system_score_codex":0.0010614728,"about_ca_system_score_gemma":0.0023655025,"threshold_uncertainty_score":0.042956054},"labels":[],"label_agreement":null},{"id":"W2946989286","doi":"10.1007/s41237-019-00085-5","title":"A concept analysis of methodological research on composite-based structural equation modeling: bridging PLSPM and GSCA","year":2019,"lang":"en","type":"article","venue":"Behaviormetrika","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":191,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Structural equation modeling; Path analysis (statistics); Computer science; Bridging (networking); Econometrics; Mathematics; Machine learning","score_opus":0.33215059425669685,"score_gpt":0.48020279125659415,"score_spread":0.1480521969998973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946989286","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026809048,0.0014097817,0.9537502,0.0046121064,0.00022999727,0.00061263266,0.0003028087,0.00018586885,0.012087551],"genre_scores_gemma":[0.3208615,0.00094233075,0.6733228,0.0008927,0.00015931566,0.0022080927,0.00027365205,0.00011482817,0.0012248559],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.96180516,0.028912587,0.0014807035,0.0031169676,0.0042180233,0.00046662593],"domain_scores_gemma":[0.855663,0.121732436,0.004163746,0.007173066,0.010450345,0.00081742584],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.046486486,0.0012556207,0.0012985633,0.011613164,0.0032111302,0.0066948077,0.0024671932,0.0015888088,0.00697458],"category_scores_gemma":[0.10672595,0.00077717716,0.0023597875,0.011162842,0.010793316,0.0108018005,0.004519018,0.004023172,0.0005498477],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003518818,0.00007189994,0.0023704604,0.00043044236,0.00008256981,0.000037383,0.002713644,0.001107296,0.00037679326,0.9534822,0.0009508886,0.03834114],"study_design_scores_gemma":[0.00006819316,0.00019119863,0.0051312987,0.0009528486,0.00020977185,0.0001834191,0.005243569,0.036881115,0.0015071444,0.9330151,0.016555853,0.00006055175],"about_ca_topic_score_codex":0.002793974,"about_ca_topic_score_gemma":0.0027632497,"teacher_disagreement_score":0.9535135,"about_ca_system_score_codex":0.005687228,"about_ca_system_score_gemma":0.0107701095,"threshold_uncertainty_score":0.24584693},"labels":[],"label_agreement":null},{"id":"W2949826916","doi":"10.48550/arxiv.1810.06387","title":"I can see clearly now: reinterpreting statistical significance","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Null hypothesis; Statistical significance; Null (SQL); CLARITY; Statistical hypothesis testing; Alternative hypothesis; Context (archaeology); Significance testing; Psychology; Cognitive psychology; Computer science; Econometrics; Mathematics; Statistics; History; Biology; Data mining","score_opus":0.041688840126429684,"score_gpt":0.21509957104833752,"score_spread":0.17341073092190784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2949826916","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005193828,0.019278863,0.4368737,0.45440623,0.06295582,0.00014475772,0.00056411844,0.0018932318,0.018689474],"genre_scores_gemma":[0.27415875,0.018444253,0.34445027,0.27589428,0.06874004,0.0011786286,0.00064169883,0.004740931,0.011751252],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.8102058,0.15186216,0.008472889,0.009125874,0.019070454,0.0012627167],"domain_scores_gemma":[0.44270378,0.47666478,0.017978752,0.035649896,0.022542238,0.004460644],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.13471563,0.001899711,0.0025452662,0.0057606753,0.0040395046,0.018799258,0.005753723,0.008087531,0.007835556],"category_scores_gemma":[0.4582656,0.0012614105,0.001640822,0.0046779956,0.055906866,0.02681846,0.010630617,0.03477624,0.0069291852],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019782197,0.00004165551,0.0017154607,0.0010985284,0.00023540742,0.000553644,0.018708108,0.00070925953,0.0017739219,0.73472846,0.17345089,0.066786855],"study_design_scores_gemma":[0.000046438505,0.00009434145,0.0008411674,0.0010848227,0.00008735865,0.00070342346,0.0035205823,0.002221509,0.0014075032,0.7435205,0.24634035,0.00013209405],"about_ca_topic_score_codex":0.0015188407,"about_ca_topic_score_gemma":0.001222893,"teacher_disagreement_score":0.8652844,"about_ca_system_score_codex":0.0032217337,"about_ca_system_score_gemma":0.005301872,"threshold_uncertainty_score":0.7124529},"labels":[],"label_agreement":null},{"id":"W2950281387","doi":"10.48550/arxiv.1307.8060","title":"Extracting Information-rich Part of Texts using Text Denoising","year":2013,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Readability; Computer science; Search engine indexing; Set (abstract data type); Text processing; Information retrieval; Relation (database); Noise reduction; Natural language processing; Artificial intelligence; Domain (mathematical analysis); Rest (music); Information extraction; Data mining; Mathematics","score_opus":0.07351421345146757,"score_gpt":0.21806988502735455,"score_spread":0.144555671575887,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950281387","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1016151,0.0026415396,0.8844853,0.0008119972,0.00052130333,0.0002482635,0.0022879255,0.0037629425,0.0036256434],"genre_scores_gemma":[0.30192703,0.0022810139,0.67330974,0.00046729506,0.0012525606,0.00036828496,0.009318233,0.0010238615,0.010051878],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986338,0.00023288316,0.00014123219,0.00034812017,0.0005410891,0.00010296764],"domain_scores_gemma":[0.99494165,0.0025513545,0.0005748093,0.00082307804,0.0009704539,0.00013864654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012620054,0.0013832148,0.0013566944,0.0050538494,0.0006786347,0.001915931,0.0007873872,0.0010911942,0.0026112674],"category_scores_gemma":[0.007343059,0.00040160315,0.0010771726,0.0029717784,0.0011654916,0.002419729,0.0012257268,0.001537423,0.0036556982],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010410701,0.00020827062,0.0017822997,0.00088411494,0.00014673105,0.0009464093,0.0007585642,0.013058724,0.4013578,0.0063858386,0.008032919,0.56539726],"study_design_scores_gemma":[0.00013295072,0.0007716011,0.012869094,0.00018394794,0.00044879198,0.0022501347,0.0010315697,0.3671386,0.50771356,0.0438172,0.06342566,0.00021694909],"about_ca_topic_score_codex":0.0006551776,"about_ca_topic_score_gemma":0.00080418325,"teacher_disagreement_score":0.0050538494,"about_ca_system_score_codex":0.00034066246,"about_ca_system_score_gemma":0.0006238409,"threshold_uncertainty_score":0.008735597},"labels":[],"label_agreement":null},{"id":"W2950657507","doi":"10.1073/pnas.1914370116","title":"Predicting research trends with semantic and neural networks with an application in quantum physics","year":2020,"lang":"en","type":"article","venue":"Proceedings of the National Academy of Sciences","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":97,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"Universität Wien; Austrian Science Fund","keywords":"Artificial neural network; Quantum; Computer science; Physics; Artificial intelligence; Quantum mechanics","score_opus":0.06736840998995927,"score_gpt":0.3578810530827379,"score_spread":0.2905126430927786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950657507","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43562242,0.0037547713,0.53783035,0.0074100415,0.00036455816,0.0001804834,0.004554681,0.0016156202,0.008667121],"genre_scores_gemma":[0.8386218,0.0012142323,0.15482289,0.00039597138,0.00025637678,0.00018384102,0.0029468895,0.000066320696,0.0014918152],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993187,0.0002820104,0.000049200542,0.00017455453,0.000121347744,0.000054096243],"domain_scores_gemma":[0.99495584,0.0033986778,0.00069873506,0.00029160854,0.0005157142,0.00013939894],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0021470503,0.00068880164,0.00042896162,0.0056832815,0.0008631969,0.0013079197,0.00058307615,0.0012628916,0.0012863808],"category_scores_gemma":[0.011370121,0.0003054651,0.00084083254,0.00485237,0.0008777213,0.002627171,0.0010007995,0.0011294946,0.00024132665],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005072645,0.00053157314,0.05997109,0.00044519216,0.0004505233,0.0003791533,0.00050835294,0.6489235,0.0035115662,0.05257102,0.0077359076,0.22446494],"study_design_scores_gemma":[0.000017548464,0.00002852801,0.0038926925,0.00003071993,0.000038050573,0.000037024154,0.000058171612,0.9409529,0.00052818895,0.052912146,0.0014886813,0.000015431195],"about_ca_topic_score_codex":0.0066562737,"about_ca_topic_score_gemma":0.011790876,"teacher_disagreement_score":0.9943167,"about_ca_system_score_codex":0.0014033233,"about_ca_system_score_gemma":0.0007833279,"threshold_uncertainty_score":0.013235033},"labels":[],"label_agreement":null},{"id":"W2950905275","doi":"10.48550/arxiv.1706.06542","title":"Extract with Order for Coherent Multi-Document Summarization","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Automatic summarization; Readability; Coherence (philosophical gambling strategy); Computer science; Rank (graph theory); Natural language processing; Key (lock); Selection (genetic algorithm); Sentence; Artificial intelligence; Information retrieval; Mathematics","score_opus":0.08428641476666028,"score_gpt":0.23998538557968446,"score_spread":0.1556989708130242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950905275","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016338125,0.0019254055,0.97260475,0.0003589347,0.00023244864,0.00014809868,0.0012069005,0.0058636214,0.001321739],"genre_scores_gemma":[0.20669737,0.0013031227,0.77346,0.000333123,0.00064824946,0.0002969941,0.007877911,0.0008510822,0.008532101],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987256,0.00030854045,0.00014636166,0.00032816693,0.00039919425,0.000092132985],"domain_scores_gemma":[0.9969331,0.0011946077,0.00036578547,0.0006151484,0.00077695766,0.00011436761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014311048,0.0013849159,0.0010836915,0.002727507,0.0005467263,0.0018582181,0.0008177322,0.0008720555,0.0037345632],"category_scores_gemma":[0.0058346977,0.00044587778,0.0008704404,0.0018046828,0.00038732192,0.0026225583,0.0011484447,0.0015450191,0.003723599],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005408091,0.00020999114,0.0011986378,0.00074771856,0.00019552672,0.00028761927,0.0005115726,0.023990486,0.058245737,0.011726873,0.024707068,0.8776379],"study_design_scores_gemma":[0.00019019496,0.0013332291,0.0037434332,0.00016431317,0.00055747177,0.0007071363,0.00060425047,0.75622463,0.10911074,0.063352786,0.06386386,0.00014789955],"about_ca_topic_score_codex":0.00092657324,"about_ca_topic_score_gemma":0.0026592403,"teacher_disagreement_score":0.0037345632,"about_ca_system_score_codex":0.00047761088,"about_ca_system_score_gemma":0.00096674135,"threshold_uncertainty_score":0.012493372},"labels":[],"label_agreement":null},{"id":"W2950942184","doi":"10.48550/arxiv.1604.02580","title":"On the Composition of Scientific Abstracts","year":2016,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"Université du Québec à Montréal","keywords":"Sentence; Relevance (law); Similarity (geometry); Computer science; Information retrieval; Composition (language); Phenomenon; Natural language processing; Content (measure theory); Linguistics; Artificial intelligence; Mathematics; Epistemology; Philosophy","score_opus":0.060575209161011875,"score_gpt":0.20249354508480127,"score_spread":0.14191833592378938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950942184","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.655599,0.014235893,0.26370275,0.0041975095,0.0017775254,0.0020667615,0.0128654465,0.002929348,0.042625804],"genre_scores_gemma":[0.7340947,0.0042289067,0.23364346,0.0005772433,0.001513847,0.0011894061,0.017786806,0.0009557734,0.006009847],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.97078615,0.009240532,0.0056572654,0.003465097,0.010306288,0.00054460444],"domain_scores_gemma":[0.81471854,0.096731246,0.036596056,0.009527781,0.039522525,0.0029037672],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.014373895,0.0007560015,0.0010941754,0.019407878,0.0031586466,0.006232289,0.001075343,0.0011919809,0.0040202667],"category_scores_gemma":[0.15607949,0.00069043064,0.00087597006,0.01769636,0.0019139341,0.008166995,0.0033380142,0.0012223498,0.0019558438],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031783958,0.00036963326,0.10938793,0.008332234,0.0009146906,0.0029758639,0.027115481,0.009743929,0.07495951,0.07560669,0.026524188,0.6608915],"study_design_scores_gemma":[0.00033183963,0.0014401211,0.28631493,0.0022283813,0.0016315053,0.008074673,0.018275313,0.091591656,0.06326738,0.26508316,0.2611853,0.0005757669],"about_ca_topic_score_codex":0.0012795889,"about_ca_topic_score_gemma":0.0010760188,"teacher_disagreement_score":0.9856261,"about_ca_system_score_codex":0.0017028442,"about_ca_system_score_gemma":0.0026910885,"threshold_uncertainty_score":0.07601732},"labels":[],"label_agreement":null},{"id":"W2950982165","doi":"","title":"Coherent Keyphrase Extraction via Web Mining","year":2003,"lang":"en","type":"preprint","venue":"NPARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Information retrieval; Task (project management); Cluster analysis; Search engine indexing; Natural language processing; Artificial intelligence","score_opus":0.021646781211188282,"score_gpt":0.30370065186152373,"score_spread":0.28205387065033544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950982165","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02293778,0.001704233,0.94419354,0.00043248135,0.00016250781,0.0006867388,0.00344832,0.021715147,0.004719351],"genre_scores_gemma":[0.09711329,0.0010340214,0.8848361,0.00019304082,0.00013504867,0.00031590305,0.0102611305,0.0009857817,0.0051256255],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99703836,0.00042232533,0.0003738699,0.0009199655,0.0010160665,0.00022930816],"domain_scores_gemma":[0.994712,0.0018479648,0.00059915095,0.0011678122,0.0015081592,0.00016485863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018544577,0.0017571027,0.0013252363,0.011698345,0.0012881807,0.0033265122,0.0021266665,0.0015291084,0.0059715295],"category_scores_gemma":[0.011411306,0.0009180094,0.0019449474,0.009417628,0.00079318695,0.0052100425,0.0026214945,0.0015078441,0.00962709],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033180552,0.00027623327,0.0033552793,0.00093970774,0.00015971184,0.0007349236,0.00043888186,0.0063271355,0.040209148,0.009238003,0.021019366,0.9169698],"study_design_scores_gemma":[0.00021990722,0.00033927432,0.0077226763,0.00034990886,0.0003281633,0.0027408826,0.0013933148,0.5796871,0.15028168,0.09044608,0.16624287,0.00024818053],"about_ca_topic_score_codex":0.0029020302,"about_ca_topic_score_gemma":0.0045755147,"teacher_disagreement_score":0.011698345,"about_ca_system_score_codex":0.0008847492,"about_ca_system_score_gemma":0.002123325,"threshold_uncertainty_score":0.019976735},"labels":[],"label_agreement":null},{"id":"W2951232267","doi":"10.48550/arxiv.1211.6321","title":"Citation content analysis (cca): A framework for syntactic and semantic analysis of citation content","year":2012,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Joint Information Systems Committee; National Science Foundation","keywords":"Citation; Computer science; Scope (computer science); Content analysis; Semantic analysis (machine learning); Citation analysis; Content (measure theory); Context (archaeology); Information retrieval; Ranking (information retrieval); Data science; World Wide Web; Sociology; Social science; Mathematics","score_opus":0.17887892196990188,"score_gpt":0.25577303882443764,"score_spread":0.07689411685453576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2951232267","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036155374,0.003019493,0.9759346,0.0018421496,0.00038419882,0.00077895925,0.0019518102,0.0016314012,0.010841777],"genre_scores_gemma":[0.084814586,0.0032656458,0.9000776,0.0004603959,0.0011542672,0.0030512277,0.0028643517,0.001030037,0.0032818995],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.95746607,0.022984378,0.004211924,0.004000421,0.010563424,0.0007737345],"domain_scores_gemma":[0.88181555,0.083628416,0.008143491,0.011225832,0.014032299,0.0011544166],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.03428156,0.002767688,0.0028343066,0.069572106,0.0064389445,0.017808326,0.0043558674,0.0033877173,0.0064999606],"category_scores_gemma":[0.10240364,0.0010637788,0.0038853735,0.0552153,0.009635637,0.021995846,0.0073602074,0.004185991,0.003111575],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048218364,0.00005042804,0.0026457277,0.0016786798,0.0003324847,0.00018493668,0.005957067,0.003034991,0.0019754726,0.8057012,0.010405492,0.16798536],"study_design_scores_gemma":[0.000029197792,0.00004505402,0.002750319,0.00092023605,0.00023638835,0.0004226067,0.0020952928,0.018669114,0.0031215765,0.8412643,0.13021621,0.00022981966],"about_ca_topic_score_codex":0.0059830123,"about_ca_topic_score_gemma":0.003638959,"teacher_disagreement_score":0.96571845,"about_ca_system_score_codex":0.007076063,"about_ca_system_score_gemma":0.012033009,"threshold_uncertainty_score":0.1813004},"labels":[],"label_agreement":null},{"id":"W2951739148","doi":"10.48550/arxiv.1204.2231","title":"Investigating Keyphrase Indexing with Text Denoising","year":2012,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Search engine indexing; Computer science; Benchmark (surveying); Noise reduction; Artificial intelligence; Natural language processing; Noise (video); Information retrieval; Energy (signal processing); Pattern recognition (psychology); Mathematics; Image (mathematics); Statistics","score_opus":0.07049531634463828,"score_gpt":0.20290324027272943,"score_spread":0.13240792392809114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2951739148","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4796156,0.015251303,0.4461291,0.0015830136,0.001172299,0.00086317374,0.0047380277,0.034425564,0.016221892],"genre_scores_gemma":[0.49500284,0.0026283357,0.46916142,0.00061828067,0.00075960666,0.00044469238,0.017673604,0.0014671574,0.012244055],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99566686,0.0010919225,0.0004225921,0.001020312,0.0015124134,0.00028597543],"domain_scores_gemma":[0.9832253,0.010049519,0.0010749507,0.0031566457,0.0020910266,0.00040262315],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006227141,0.0017334941,0.0018624531,0.0049908673,0.0010416028,0.0030477378,0.0020774463,0.0018351978,0.0028492797],"category_scores_gemma":[0.029901518,0.0004245852,0.0011206933,0.0045027253,0.0016935864,0.007293322,0.0021206208,0.001989883,0.0040295203],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001906081,0.00065585016,0.006889872,0.002030083,0.00034794674,0.00035387668,0.0011299588,0.026876878,0.13752624,0.004831741,0.015194631,0.80225676],"study_design_scores_gemma":[0.00025883943,0.0020609703,0.013402975,0.00014712571,0.0003419595,0.0013347077,0.0013311243,0.563528,0.36289608,0.010608358,0.043836996,0.0002528338],"about_ca_topic_score_codex":0.003931822,"about_ca_topic_score_gemma":0.004606718,"teacher_disagreement_score":0.006227141,"about_ca_system_score_codex":0.0010849615,"about_ca_system_score_gemma":0.0012440059,"threshold_uncertainty_score":0.03293264},"labels":[],"label_agreement":null},{"id":"W2953722276","doi":"10.1016/j.ipm.2019.102063","title":"A multi-centrality index for graph-based keyword extraction","year":2019,"lang":"en","type":"article","venue":"Information Processing & Management","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":92,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Fundação de Amparo à Pesquisa do Estado de São Paulo","keywords":"Centrality; Betweenness centrality; PageRank; Computer science; Clustering coefficient; Cluster analysis; Graph; Artificial intelligence; Data mining; Natural language processing; Information retrieval; Theoretical computer science; Mathematics; Statistics","score_opus":0.01397295543473869,"score_gpt":0.2989504196007305,"score_spread":0.2849774641659918,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953722276","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.068713225,0.0028866003,0.9058158,0.00038621918,0.00034325564,0.0004578163,0.009036384,0.0070596747,0.005301059],"genre_scores_gemma":[0.33186582,0.0011668465,0.64572555,0.00012967286,0.00037081316,0.0004263093,0.01443012,0.0006524317,0.005232514],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982261,0.00020545887,0.00020768438,0.00034825478,0.00087384373,0.00013873038],"domain_scores_gemma":[0.9964018,0.0012465012,0.0003928252,0.00033504795,0.0014121876,0.00021172625],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000880022,0.000815603,0.0013299121,0.01497734,0.0013797679,0.002121515,0.0012293531,0.0009950873,0.0034783697],"category_scores_gemma":[0.0066680624,0.00035324943,0.00093238143,0.012154982,0.00038932412,0.0027962774,0.0013467484,0.0006711809,0.0025842402],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008157961,0.00042693043,0.013810256,0.0011312519,0.00046022027,0.00057274086,0.00040770785,0.025894504,0.08161053,0.018034728,0.035626836,0.8212084],"study_design_scores_gemma":[0.00014312382,0.00036191323,0.014628559,0.00014781431,0.00045077503,0.0017783694,0.00043420598,0.8581312,0.043070644,0.040381446,0.04026389,0.00020820362],"about_ca_topic_score_codex":0.0054911,"about_ca_topic_score_gemma":0.011454771,"teacher_disagreement_score":0.01497734,"about_ca_system_score_codex":0.00095428975,"about_ca_system_score_gemma":0.0017029275,"threshold_uncertainty_score":0.011636317},"labels":[],"label_agreement":null},{"id":"W2954858138","doi":"10.22260/isarc2019/0171","title":"Automatic Key-phrase Extraction to Support the Understanding of Infrastructure Disaster Resilience","year":2019,"lang":"en","type":"article","venue":"Proceedings of the ... ISARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Resilience (materials science); Computer science; Key (lock); Phrase; Critical infrastructure; Computer security; Natural language processing","score_opus":0.013635449803248044,"score_gpt":0.27092614887838457,"score_spread":0.2572906990751365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2954858138","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044410545,0.0047731004,0.87237346,0.0018567559,0.00057661947,0.0024329661,0.028691147,0.03480472,0.010080726],"genre_scores_gemma":[0.10830669,0.0021516979,0.8436206,0.0003410585,0.00026513627,0.0009350917,0.039337195,0.0008586319,0.0041839276],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983028,0.00040122666,0.00028725693,0.00040768128,0.00046170643,0.00013919735],"domain_scores_gemma":[0.99407756,0.002833292,0.0006381878,0.00039015748,0.0019290661,0.00013169761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017369121,0.0024350907,0.0011558176,0.00920075,0.0010629102,0.0021586632,0.0013135787,0.0013717461,0.010505778],"category_scores_gemma":[0.007966606,0.00056456414,0.0014068459,0.0050544296,0.00062136113,0.004311051,0.0023370907,0.0014880458,0.0097518535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068076525,0.00026205284,0.003289579,0.004802914,0.00019588796,0.0017240188,0.0019033895,0.0026369076,0.14865716,0.0070903753,0.07992647,0.7488305],"study_design_scores_gemma":[0.00042175627,0.00088929577,0.02665412,0.0013860901,0.0007645429,0.0059722345,0.0076722386,0.2693127,0.2501653,0.03476659,0.40151903,0.00047608468],"about_ca_topic_score_codex":0.0036057525,"about_ca_topic_score_gemma":0.0035399646,"teacher_disagreement_score":0.010505778,"about_ca_system_score_codex":0.0010423433,"about_ca_system_score_gemma":0.002294714,"threshold_uncertainty_score":0.035145283},"labels":[],"label_agreement":null},{"id":"W2955707177","doi":"10.3758/s13428-019-01268-4","title":"The Semantic Librarian: A search engine built from vector-space models of semantics","year":2019,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; University of Winnipeg","funders":"","keywords":"Computer science; Semantics (computer science); Space (punctuation); Cognition; Semantic space; Information retrieval; Cognitive science; Human–computer interaction; Artificial intelligence; World Wide Web; Data science; Psychology; Programming language","score_opus":0.1924456178434242,"score_gpt":0.5118749448262684,"score_spread":0.3194293269828442,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955707177","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015124008,0.00080131035,0.86747366,0.00049633265,0.00007938323,0.00031174635,0.017695254,0.09160772,0.006410657],"genre_scores_gemma":[0.28188378,0.0015578207,0.67131746,0.0004895838,0.00007184114,0.0006348941,0.030296745,0.0058867955,0.007861065],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997154,0.000067771056,0.00002860514,0.00007248868,0.000095631134,0.0000201217],"domain_scores_gemma":[0.99927145,0.00042768798,0.000043795288,0.000114893504,0.00010016269,0.00004190873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005917797,0.00082810124,0.0009952191,0.0027915426,0.0003569002,0.0016921777,0.0012225441,0.0008928546,0.010437401],"category_scores_gemma":[0.0049169315,0.00047808484,0.000982731,0.002498026,0.00030413913,0.0040473044,0.0013035552,0.0006345,0.003775123],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013187716,0.0004596188,0.008665559,0.0015839295,0.0006541116,0.0004619866,0.00068274816,0.11150115,0.0066500735,0.1320523,0.14731123,0.58865845],"study_design_scores_gemma":[0.000112335496,0.00009034135,0.0010498497,0.000085985426,0.00012931967,0.00020381816,0.00009020928,0.8636282,0.003398329,0.09631175,0.034845203,0.000054693097],"about_ca_topic_score_codex":0.00767375,"about_ca_topic_score_gemma":0.017463198,"teacher_disagreement_score":0.010437401,"about_ca_system_score_codex":0.0007149639,"about_ca_system_score_gemma":0.0013900403,"threshold_uncertainty_score":0.03491658},"labels":[],"label_agreement":null},{"id":"W2960010094","doi":"10.1007/978-3-030-47124-8_44","title":"Analysis of Word Embeddings Using Fuzzy Clustering","year":2020,"lang":"en","type":"book-chapter","venue":"Studies in fuzziness and soft computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Word (group theory); Fuzzy clustering; Cluster analysis; Artificial intelligence; Fuzzy logic; Computer science; Representation (politics); Data mining; Curse of dimensionality; Pattern recognition (psychology); Similarity (geometry); Mathematics; Natural language processing","score_opus":0.06611174490473551,"score_gpt":0.34387517022084824,"score_spread":0.27776342531611276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2960010094","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17405862,0.0016019901,0.8160491,0.00034330218,0.00025274992,0.0001244523,0.0018088692,0.0018603456,0.0039004788],"genre_scores_gemma":[0.44744092,0.00075176987,0.5410014,0.00006117312,0.00009071211,0.00012355401,0.004963816,0.00038704398,0.0051796976],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992902,0.00015136405,0.00007885067,0.00018067853,0.00024424246,0.000054594202],"domain_scores_gemma":[0.99824786,0.0008095424,0.00012435412,0.00018115327,0.00058753055,0.000049656832],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00061140297,0.00064348476,0.000616921,0.0034107508,0.0006269687,0.0017053466,0.0006407166,0.00055803725,0.002951897],"category_scores_gemma":[0.004182546,0.00023708648,0.00087657874,0.0040771407,0.00041545005,0.0021360454,0.00081939076,0.00078110286,0.0013911779],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004170448,0.00016596683,0.0061341184,0.0005390367,0.00022258623,0.00025211263,0.00082772935,0.03184265,0.027511084,0.022238256,0.0073767817,0.9024727],"study_design_scores_gemma":[0.000029647716,0.00015084766,0.010442707,0.000117736,0.00018345025,0.00053035875,0.0013066523,0.88336265,0.025418084,0.066396244,0.011970442,0.000091158414],"about_ca_topic_score_codex":0.0025928852,"about_ca_topic_score_gemma":0.0029226064,"teacher_disagreement_score":0.0034107508,"about_ca_system_score_codex":0.0005052165,"about_ca_system_score_gemma":0.0005779661,"threshold_uncertainty_score":0.009875059},"labels":[],"label_agreement":null},{"id":"W2961582678","doi":"10.1016/j.prevetmed.2019.104728","title":"The inappropriate use of formulae and references and the possible domino effect of spurious results","year":2019,"lang":"en","type":"letter","venue":"Preventive Veterinary Medicine","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Food Inspection Agency","funders":"Canadian Food Inspection Agency","keywords":"Spurious relationship; Domino; Domino effect; Econometrics; Statistics; Mathematics; Computer science; Biology; Physics","score_opus":0.029180328592751305,"score_gpt":0.2981293379953577,"score_spread":0.26894900940260635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2961582678","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005225505,0.00097079185,0.00235502,0.9715256,0.02071653,0.000024220086,0.00007026437,0.00009670882,0.003718334],"genre_scores_gemma":[0.010291961,0.0011740949,0.007487757,0.92662776,0.04336227,0.00009731313,0.00004912059,0.00020921262,0.010700516],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.96060765,0.01787418,0.006301389,0.0022042491,0.0117966235,0.0012159387],"domain_scores_gemma":[0.7010619,0.2335864,0.012120884,0.0076453723,0.041664008,0.0039214566],"candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.026764985,0.0006494054,0.001570404,0.0022104303,0.0025789605,0.004838395,0.0031376043,0.02949921,0.007748817],"category_scores_gemma":[0.2586591,0.00082944497,0.0011155471,0.0018805224,0.0064159385,0.0050443057,0.0019270601,0.0295601,0.010096323],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016056185,0.000047860336,0.0012635111,0.00029545306,0.000051738803,0.004304964,0.0009393043,0.00014842238,0.0006237859,0.016772857,0.9332947,0.04209667],"study_design_scores_gemma":[0.00010638557,0.000082947874,0.0018170213,0.00092644396,0.00007835318,0.008231591,0.00073797954,0.0023272482,0.0013534672,0.056078997,0.92814344,0.00011618378],"about_ca_topic_score_codex":0.0027093366,"about_ca_topic_score_gemma":0.0045219837,"teacher_disagreement_score":0.973235,"about_ca_system_score_codex":0.0041050143,"about_ca_system_score_gemma":0.0035155853,"threshold_uncertainty_score":0.14154845},"labels":[],"label_agreement":null},{"id":"W2962704246","doi":"10.18653/v1/p18-1062","title":"Unsupervised Abstractive Meeting Summarization with Multi-Sentence Compression and Budgeted Submodular Maximization","year":2018,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":100,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Automatic summarization; Computer science; Zhàng; Submodular set function; Natural language processing; Sentence; Maximization; Artificial intelligence; Linguistics; Volume (thermodynamics); Mathematics; Philosophy; History; Combinatorics","score_opus":0.013225451306857263,"score_gpt":0.2521054262172721,"score_spread":0.23887997491041485,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962704246","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013757573,0.0024872303,0.97318304,0.00072664267,0.00033855488,0.00023959127,0.0019168829,0.0050999722,0.002250447],"genre_scores_gemma":[0.21311444,0.001193421,0.75040823,0.00052404264,0.0012119962,0.000729747,0.018805927,0.001265491,0.0127467215],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99846053,0.00053873606,0.00013489598,0.00043833719,0.00027046885,0.00015716151],"domain_scores_gemma":[0.9978661,0.0010659867,0.00018612052,0.00030539674,0.0004700793,0.00010629239],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018549275,0.0026056168,0.0027589642,0.0021309361,0.000924187,0.0017718788,0.0025036917,0.0017025815,0.005144968],"category_scores_gemma":[0.004487707,0.00084375514,0.0014918504,0.002816517,0.0005595272,0.0032733777,0.002014599,0.0019786044,0.004284915],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006827464,0.0003312799,0.0008670987,0.00062758895,0.00031479797,0.00028494044,0.00042533287,0.08520941,0.016655974,0.006515551,0.04947925,0.83860606],"study_design_scores_gemma":[0.000086795444,0.00019140041,0.0004950899,0.000036445097,0.00014777348,0.0000993208,0.00022237991,0.97051525,0.0067487536,0.014134587,0.0072904583,0.00003173182],"about_ca_topic_score_codex":0.0038937056,"about_ca_topic_score_gemma":0.009568098,"teacher_disagreement_score":0.005144968,"about_ca_system_score_codex":0.00080582517,"about_ca_system_score_gemma":0.0016257268,"threshold_uncertainty_score":0.017211556},"labels":[],"label_agreement":null},{"id":"W2963665652","doi":"10.3390/bdcc3030044","title":"Archetype-Based Modeling and Search of Social Media","year":2019,"lang":"en","type":"article","venue":"Big Data and Cognitive Computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Compute Canada","keywords":"Archetype; Jargon; Social media; Vocabulary; Computer science; Set (abstract data type); Relevance (law); Information retrieval; Slang; Data science; World Wide Web; Linguistics","score_opus":0.11721073622008192,"score_gpt":0.3397729561917595,"score_spread":0.22256221997167758,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963665652","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.063495144,0.0007718918,0.9242205,0.0012566177,0.00005018491,0.00035585393,0.002218334,0.0011179799,0.0065134116],"genre_scores_gemma":[0.57078886,0.0008877563,0.41792262,0.00024497154,0.000085582884,0.0008183922,0.0034526845,0.0001934557,0.005605661],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99751544,0.0011028665,0.00022877424,0.00053966907,0.00049213803,0.00012108673],"domain_scores_gemma":[0.9928229,0.005171692,0.0005679095,0.00074669207,0.0005581718,0.00013262901],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002275078,0.0007137834,0.0008078072,0.0046800496,0.0010592787,0.0026402674,0.0017713627,0.0013551823,0.002306438],"category_scores_gemma":[0.012343446,0.0005507678,0.001799692,0.0037270605,0.0011380359,0.005985128,0.0019695875,0.0013041464,0.000966904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000579558,0.0004034462,0.019451978,0.0012088447,0.00035873684,0.00075338065,0.0057943147,0.4253913,0.012121684,0.2885569,0.009491614,0.23588823],"study_design_scores_gemma":[0.000012343362,0.000025999301,0.0005544587,0.000021344957,0.000018097531,0.00011828936,0.0002350216,0.94445515,0.001149194,0.05045179,0.002941486,0.000016825245],"about_ca_topic_score_codex":0.012590425,"about_ca_topic_score_gemma":0.015819913,"teacher_disagreement_score":0.012590425,"about_ca_system_score_codex":0.0016176938,"about_ca_system_score_gemma":0.0019087647,"threshold_uncertainty_score":0.025034308},"labels":[],"label_agreement":null},{"id":"W2964059043","doi":"","title":"Understanding Electric Current Using Agent-based Models: Connecting the Micro-level with Flow Rate.","year":2016,"lang":"en","type":"article","venue":"International Conference on Computer Supported Education","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Electricity; Process (computing); Current (fluid); Bootstrapping (finance); Set (abstract data type); Transient (computer programming); Electrical engineering; Engineering","score_opus":0.2443541035928746,"score_gpt":0.35301889491390503,"score_spread":0.10866479132103044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964059043","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060268007,0.00062087673,0.9120356,0.0034443927,0.00010438632,0.00015777686,0.00011183703,0.00025555454,0.023001531],"genre_scores_gemma":[0.7323609,0.0007704515,0.25963438,0.00026518342,0.000046052755,0.0004225859,0.000112594986,0.00009411423,0.006293804],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996793,0.00015947808,0.000018169987,0.000056511988,0.00006425069,0.000022360062],"domain_scores_gemma":[0.99831945,0.0011340368,0.00017949106,0.0001382292,0.0001319542,0.00009678857],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078784913,0.0005891712,0.00039017215,0.0006462261,0.0006441821,0.0026700802,0.001403177,0.001524094,0.003549049],"category_scores_gemma":[0.004199946,0.0005058414,0.0009199261,0.00030807214,0.0017341982,0.0045370683,0.0011306233,0.0012728312,0.00038626147],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043573855,0.00011889509,0.0033080748,0.0001826144,0.00007548816,0.0002478971,0.002366008,0.43565676,0.0029228646,0.5330377,0.0012152364,0.020824803],"study_design_scores_gemma":[0.000021243406,0.000031372285,0.0003349365,0.000038826107,0.000027653221,0.00005651973,0.00029449956,0.8107917,0.0006540189,0.18078814,0.0069428193,0.000018191547],"about_ca_topic_score_codex":0.0037477913,"about_ca_topic_score_gemma":0.0037509617,"teacher_disagreement_score":0.0037477913,"about_ca_system_score_codex":0.001217328,"about_ca_system_score_gemma":0.0010789344,"threshold_uncertainty_score":0.011872709},"labels":[],"label_agreement":null},{"id":"W2964298985","doi":"","title":"Automatic Text Summarization Approaches to Speed up Topic Model Learning Process","year":2016,"lang":"en","type":"other","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Automatic summarization; Computer science; Representation (politics); Information retrieval; Process (computing); Context (archaeology); Big data; Text processing; Space (punctuation); The Internet; Natural language processing; Data science; World Wide Web; Data mining","score_opus":0.02694046449698274,"score_gpt":0.25788570660585497,"score_spread":0.23094524210887224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964298985","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0142757585,0.003182691,0.9111458,0.000902486,0.0012237453,0.0006213447,0.0081438245,0.05329598,0.0072084633],"genre_scores_gemma":[0.08589793,0.0015395078,0.8491089,0.00023453856,0.0009650201,0.0006632897,0.034489114,0.002912819,0.024188893],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998665,0.00040316337,0.00015750808,0.00032855693,0.00034955528,0.000096317846],"domain_scores_gemma":[0.9955106,0.0018924322,0.00021553304,0.000492753,0.0017320012,0.00015674156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015859613,0.0020961761,0.0014460359,0.0047240513,0.0010488229,0.0024998018,0.001473059,0.0012065133,0.028814808],"category_scores_gemma":[0.0071465163,0.0006878374,0.0015873845,0.0045595467,0.0002701127,0.0034051344,0.0015106617,0.0019009898,0.024528323],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044208966,0.00020685299,0.0006033895,0.0006970268,0.00014908811,0.00013965147,0.00018429618,0.008973591,0.024128024,0.0023389752,0.09413597,0.86800116],"study_design_scores_gemma":[0.00027320994,0.0004175783,0.0027090071,0.00012371781,0.00050837477,0.00042018204,0.00032365214,0.8036972,0.060740445,0.014479289,0.11620915,0.000098211705],"about_ca_topic_score_codex":0.0036830625,"about_ca_topic_score_gemma":0.0068302047,"teacher_disagreement_score":0.028814808,"about_ca_system_score_codex":0.00061310333,"about_ca_system_score_gemma":0.0012045164,"threshold_uncertainty_score":0.096395195},"labels":[],"label_agreement":null},{"id":"W2964457291","doi":"","title":"Event Detection using Images of Temporal Word Patterns.","year":2019,"lang":"en","type":"article","venue":"NPARC","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Event (particle physics); Word (group theory); Artificial intelligence; Natural language processing; Speech recognition; Linguistics","score_opus":0.011971414708768596,"score_gpt":0.27750837842581483,"score_spread":0.2655369637170462,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964457291","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40071705,0.0047949096,0.5210428,0.0009004589,0.00093454175,0.0009904982,0.028457345,0.020087289,0.022075139],"genre_scores_gemma":[0.68109596,0.0014472915,0.29190448,0.0002005949,0.00043929898,0.0003795534,0.018053584,0.00038507083,0.006094155],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994466,0.000051885814,0.000055083507,0.00020190373,0.00017054402,0.000074113275],"domain_scores_gemma":[0.9983045,0.00049887947,0.00042455635,0.00022375918,0.00044995575,0.00009844292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00044222057,0.00092088775,0.00048553472,0.005663774,0.0003364126,0.0013384944,0.00058121874,0.0008315928,0.0026349027],"category_scores_gemma":[0.0028423164,0.00021311136,0.000553113,0.0035682162,0.0003055417,0.0021341313,0.000770317,0.00056504994,0.0027264669],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015281314,0.00032991197,0.02867655,0.001074047,0.00025733025,0.0010778842,0.0005809029,0.0039487593,0.15883563,0.0031243137,0.025660532,0.774906],"study_design_scores_gemma":[0.00012586266,0.000822751,0.16241278,0.00021574587,0.0003319599,0.004300957,0.0019652757,0.45224267,0.28766114,0.012087936,0.07768011,0.00015282545],"about_ca_topic_score_codex":0.0019062635,"about_ca_topic_score_gemma":0.0034491841,"teacher_disagreement_score":0.005663774,"about_ca_system_score_codex":0.00033751476,"about_ca_system_score_gemma":0.00027677522,"threshold_uncertainty_score":0.008814633},"labels":[],"label_agreement":null},{"id":"W2964657702","doi":"10.29173/cais295","title":"Doctoral Students’ Mental Models of a Web Search Engine","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Psychology; Mental model; Humanities; Context (archaeology); Art; Geography; Cognitive science","score_opus":0.04434080710215641,"score_gpt":0.29227336408815535,"score_spread":0.24793255698599895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964657702","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9948724,0.00012957562,0.0008076877,0.00037313523,0.000010072557,0.000030182895,0.000035368674,0.000010746771,0.003730739],"genre_scores_gemma":[0.99751353,0.00018168155,0.0004742419,0.00008277977,0.000004732804,0.000028149683,0.00007708832,0.000003510701,0.0016343701],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.99741685,0.0011475001,0.00015764975,0.00017290625,0.0008700431,0.00023508389],"domain_scores_gemma":[0.97995603,0.0096460255,0.0031962611,0.0015413393,0.0029032296,0.0027570843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006346943,0.00037777616,0.00036261324,0.0014974831,0.0010518839,0.0060527637,0.0005954343,0.0007482743,0.0031746028],"category_scores_gemma":[0.02713326,0.00033380694,0.00080307556,0.0006926899,0.0016901023,0.0025024726,0.001805047,0.0014828348,0.00062026264],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006387843,0.0013731697,0.6025451,0.00037922585,0.00022816643,0.00051010586,0.2955752,0.0028440491,0.005110518,0.012174855,0.004257159,0.07436369],"study_design_scores_gemma":[0.00018353842,0.0029814618,0.50310427,0.00037477628,0.0002711026,0.001116723,0.4204277,0.016996885,0.0057471455,0.01475385,0.03367731,0.000365188],"about_ca_topic_score_codex":0.0033377714,"about_ca_topic_score_gemma":0.002703249,"teacher_disagreement_score":0.006346943,"about_ca_system_score_codex":0.0012915604,"about_ca_system_score_gemma":0.001464476,"threshold_uncertainty_score":0.033566236},"labels":[],"label_agreement":null},{"id":"W2964996848","doi":"10.35050/jipm010.2019.031","title":"Structural analyzing of “Information Science Theories’ based on co-word network analysis of articles in Web of Science database (1983-2017)","year":2022,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Word (group theory); Web of science; Information retrieval; World Wide Web; Data science; Database; Linguistics; Political science; MEDLINE; Philosophy","score_opus":0.11379743007524001,"score_gpt":0.5063494908679989,"score_spread":0.39255206079275884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964996848","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9134351,0.0055154935,0.016582998,0.0012889712,0.00015488135,0.0005740404,0.038152687,0.00032487055,0.023970941],"genre_scores_gemma":[0.95471704,0.0025575182,0.011927277,0.00006642978,0.00014120381,0.00063365593,0.027447477,0.0000715481,0.002437749],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99529994,0.00086299435,0.00091961515,0.00072228414,0.0018294418,0.00036567939],"domain_scores_gemma":[0.96767664,0.01887305,0.0063056666,0.0009977412,0.0054366384,0.0007102625],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.003187731,0.0004811227,0.0004995731,0.049618986,0.0011122169,0.0032018044,0.0005813382,0.000507459,0.003921285],"category_scores_gemma":[0.023933545,0.00019702509,0.0010456821,0.043528616,0.0006800175,0.003791335,0.0014645095,0.0005157413,0.00083248987],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003885112,0.00016485859,0.76540655,0.0047224695,0.001062161,0.0012274237,0.008112586,0.0037883013,0.0045564435,0.019907461,0.016326817,0.17433636],"study_design_scores_gemma":[0.000032248583,0.00018476616,0.87881404,0.0011786731,0.0010670321,0.001290785,0.012927556,0.02796967,0.005158558,0.0153081035,0.055960756,0.00010794132],"about_ca_topic_score_codex":0.0061497767,"about_ca_topic_score_gemma":0.008224137,"teacher_disagreement_score":0.95038104,"about_ca_system_score_codex":0.0022250027,"about_ca_system_score_gemma":0.0024333019,"threshold_uncertainty_score":0.016858518},"labels":[],"label_agreement":null},{"id":"W2965420532","doi":"10.29173/cais548","title":"Re-Conceiving Information Studies: A Quantum Approach","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Instrumentalism; The Renaissance; Humanities; Epistemology; Sociology; Philosophy; Art; Art history","score_opus":0.04289168566847006,"score_gpt":0.28216611247005735,"score_spread":0.23927442680158728,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2965420532","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017898645,0.027055133,0.5062921,0.23225278,0.0047660903,0.00022335582,0.00026000175,0.00036829364,0.21088359],"genre_scores_gemma":[0.75444084,0.01672626,0.17790209,0.013595581,0.0049684774,0.00065576733,0.00015042428,0.0004616316,0.031098971],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9912633,0.005989749,0.00033733348,0.0007155819,0.0013003845,0.00039364648],"domain_scores_gemma":[0.9796246,0.0144029325,0.0006342466,0.0028434978,0.0016986782,0.00079597766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016575782,0.0007651647,0.001272161,0.004661055,0.004535352,0.012733009,0.0032868423,0.0039499537,0.009022497],"category_scores_gemma":[0.018387115,0.00058196654,0.001714258,0.002257972,0.04882579,0.033555083,0.007192262,0.007923141,0.0010676297],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000037630061,0.0000047799595,0.000027006734,0.000025322859,0.0000023325695,0.000014546119,0.000815796,0.000091763926,0.000037953763,0.99666697,0.0005011589,0.0018087701],"study_design_scores_gemma":[0.000006126033,0.0000061445985,0.000033259035,0.00003474492,0.0000023233533,0.000017891345,0.00057540194,0.00045429743,0.00004372727,0.9860336,0.012785241,0.0000072554576],"about_ca_topic_score_codex":0.0029655995,"about_ca_topic_score_gemma":0.0026539315,"teacher_disagreement_score":0.016575782,"about_ca_system_score_codex":0.0068595116,"about_ca_system_score_gemma":0.005339519,"threshold_uncertainty_score":0.08766216},"labels":[],"label_agreement":null},{"id":"W2966026905","doi":"10.24963/ijcai.2019/712","title":"Unsupervised Neural Aspect Extraction with Sememes","year":2019,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Youth Innovation Promotion Association; National Natural Science Foundation of China; Ant Financial Services Group; Tencent; National Key Research and Development Program of China; Youth Innovation Promotion Association of the Chinese Academy of Sciences","keywords":"Computer science; Artificial intelligence; Natural language processing; Semantics (computer science); Coherence (philosophical gambling strategy); Sentence; Context (archaeology); Word (group theory); Artificial neural network; Linguistics","score_opus":0.007519555874121481,"score_gpt":0.25303840469536454,"score_spread":0.24551884882124306,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2966026905","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09978336,0.0009820708,0.8885842,0.00035025048,0.000106453896,0.00015439457,0.0007224581,0.004908107,0.0044087823],"genre_scores_gemma":[0.6624169,0.0005836791,0.32542437,0.00032798463,0.00011054602,0.00018613263,0.003010965,0.00025780062,0.007681612],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997987,0.000034698,0.000014004692,0.00009046133,0.000039925653,0.000022170365],"domain_scores_gemma":[0.9996468,0.00014599136,0.000041549563,0.00006616101,0.00008446303,0.000015018261],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003281413,0.00071847026,0.0003824233,0.0010560508,0.0002643215,0.00059391896,0.00079912314,0.0005981361,0.0014936435],"category_scores_gemma":[0.0012491512,0.00029522862,0.00077461003,0.0011152676,0.00039518392,0.001627683,0.00066327024,0.0009578317,0.0005683497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020477203,0.0001906806,0.0045838077,0.00020993175,0.00018416421,0.00037654903,0.00023624611,0.05305179,0.03995513,0.010299384,0.0075832675,0.88312423],"study_design_scores_gemma":[0.000016466336,0.000070980976,0.002605437,0.00001982578,0.00006068112,0.00013348325,0.000059072674,0.9568457,0.013467099,0.0223947,0.004309482,0.000017125903],"about_ca_topic_score_codex":0.002474953,"about_ca_topic_score_gemma":0.009106853,"teacher_disagreement_score":0.002474953,"about_ca_system_score_codex":0.0004039497,"about_ca_system_score_gemma":0.0005325122,"threshold_uncertainty_score":0.004996717},"labels":[],"label_agreement":null},{"id":"W2969275073","doi":"","title":"An Efficient Method to Determine which Combination of Keywords Triggered Automatic Filtering of a Message","year":2019,"lang":"en","type":"article","venue":"USENIX Security Symposium","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Data mining","score_opus":0.00827297475080617,"score_gpt":0.28901719527453423,"score_spread":0.28074422052372805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2969275073","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025895135,0.0004308362,0.95137715,0.00016031544,0.00017198353,0.00028611877,0.0008715065,0.018480094,0.002326987],"genre_scores_gemma":[0.23352559,0.00017907852,0.755169,0.00015210651,0.00010546354,0.00032766504,0.0015890737,0.0005703345,0.008381699],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984163,0.0001695458,0.0001513416,0.00039564492,0.0007198427,0.00014724904],"domain_scores_gemma":[0.99774593,0.0007410562,0.00024795707,0.00035891333,0.0007860409,0.00011999285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007258187,0.0011645616,0.0010495506,0.0030690024,0.00082316645,0.0017120285,0.00088193343,0.0009585612,0.0061271093],"category_scores_gemma":[0.002908137,0.00042335672,0.00055600406,0.0013217479,0.00035949663,0.0013272472,0.00084040477,0.0006686921,0.003869194],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00097180984,0.00017835826,0.004206659,0.00025266915,0.00013487399,0.00027497008,0.00016067276,0.0033144623,0.21989103,0.0050724065,0.011133831,0.7544082],"study_design_scores_gemma":[0.00022749195,0.00047448228,0.009667611,0.00006514888,0.0002966347,0.0017749824,0.00023752535,0.5270186,0.4160238,0.013376605,0.03066206,0.00017511616],"about_ca_topic_score_codex":0.00268139,"about_ca_topic_score_gemma":0.0055481778,"teacher_disagreement_score":0.0061271093,"about_ca_system_score_codex":0.00083835045,"about_ca_system_score_gemma":0.0018114987,"threshold_uncertainty_score":0.020497262},"labels":[],"label_agreement":null},{"id":"W297012442","doi":"","title":"Disseminative Characteristics of Poetry Embryology and Development/LES CARACTÉRISTIQUES COMMUNICATIVES DE LA POÉSIE DURANT SON APPARITION ET DÉVELOPPEMENT","year":2008,"lang":"fr","type":"article","venue":"Canadian social science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Poetry; Literature; Humanities; Style (visual arts); Art; Philosophy","score_opus":0.03189833034654696,"score_gpt":0.33676293756220727,"score_spread":0.30486460721566033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W297012442","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90604484,0.0028209847,0.007286192,0.0012702902,0.000114266826,0.00012657946,0.00014880238,0.00010690486,0.08208113],"genre_scores_gemma":[0.9910802,0.00059511774,0.0015355733,0.000046734564,0.000044223772,0.000049544717,0.000052272837,0.000026236776,0.0065701343],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980489,0.0009671455,0.00010050173,0.00023641875,0.0005404315,0.00010664311],"domain_scores_gemma":[0.98366815,0.009653715,0.0023292436,0.0009565753,0.002485382,0.00090689294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031370001,0.00020098941,0.00019542914,0.0020458126,0.0016873585,0.0032577608,0.00037187646,0.00063416053,0.0059004817],"category_scores_gemma":[0.021242594,0.0002065475,0.00019729203,0.001433599,0.0023974665,0.0029281976,0.0013942723,0.0011258819,0.0008231641],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007081541,0.00024233838,0.17876029,0.0006673181,0.000059968515,0.006275959,0.3949509,0.00046313615,0.053948745,0.08913198,0.0031247805,0.27166635],"study_design_scores_gemma":[0.000032663305,0.00054009835,0.6606494,0.00044047905,0.00007166939,0.00847111,0.17031887,0.0010745084,0.017274085,0.019670477,0.12130886,0.00014766412],"about_ca_topic_score_codex":0.00084086973,"about_ca_topic_score_gemma":0.0006761563,"teacher_disagreement_score":0.0059004817,"about_ca_system_score_codex":0.00096899236,"about_ca_system_score_gemma":0.0010800805,"threshold_uncertainty_score":0.019739032},"labels":[],"label_agreement":null},{"id":"W2970591538","doi":"","title":"KlickLabs at the TAC 2018 Drug-drug Interaction Extraction from Drug Labels Track.","year":2018,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Drug; Track (disk drive); Drug-drug interaction; Extraction (chemistry); Computer science; Pharmacology; Chemistry; Medicine; Chromatography","score_opus":0.007887684036896365,"score_gpt":0.28527664641641864,"score_spread":0.27738896237952226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970591538","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009395465,0.006691128,0.21600653,0.01105419,0.0076094163,0.0008938528,0.4577799,0.23890314,0.051666413],"genre_scores_gemma":[0.04336859,0.0020459737,0.2155238,0.0018647431,0.0011928267,0.0006499326,0.64086163,0.010988075,0.0835044],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964953,0.0006422261,0.00023702592,0.00088268326,0.0014430091,0.00029984838],"domain_scores_gemma":[0.9923497,0.0022634696,0.00034054465,0.00192227,0.0023461396,0.00077773235],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054807765,0.002360524,0.0021295554,0.0068969764,0.0013517245,0.004810159,0.002304306,0.0020316339,0.10049659],"category_scores_gemma":[0.016749533,0.0007678022,0.0015878653,0.004358813,0.0004437484,0.0058249366,0.0036206134,0.0023258685,0.1052355],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037573467,0.00019211048,0.00091297703,0.00046514624,0.00010984085,0.00009691462,0.00006257964,0.0011781434,0.0037457084,0.0032124778,0.8536666,0.13598178],"study_design_scores_gemma":[0.00027397275,0.00026691786,0.003393949,0.000209903,0.00015667873,0.0002026661,0.00012210161,0.06458635,0.014722791,0.023465319,0.89249206,0.000107263586],"about_ca_topic_score_codex":0.009002505,"about_ca_topic_score_gemma":0.016746635,"teacher_disagreement_score":0.10049659,"about_ca_system_score_codex":0.00169943,"about_ca_system_score_gemma":0.0026611974,"threshold_uncertainty_score":0.33619457},"labels":[],"label_agreement":null},{"id":"W2972128553","doi":"10.22215/etd/2019-13484","title":"Persuasive Content Generator The Design, Development and Validation of Persuasive Contect Generator Based on Social Media Profiles","year":2019,"lang":"en","type":"dissertation","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Popularity; Categorization; Social media; Comprehension; Internet privacy; Persuasion; World Wide Web; The Internet; Generator (circuit theory); Psychology; Social psychology; Power (physics); Artificial intelligence","score_opus":0.06364569148738479,"score_gpt":0.29159320748186235,"score_spread":0.22794751599447755,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2972128553","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12107337,0.00017407766,0.83733946,0.00032054321,0.00020341591,0.0160026,0.00057386636,0.0140197,0.01029295],"genre_scores_gemma":[0.29182357,0.00011550287,0.68944615,0.00019449882,0.000060308914,0.008460273,0.0011606996,0.0010222986,0.0077167037],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9914773,0.004522371,0.00075268256,0.0010849156,0.0019381656,0.00022460784],"domain_scores_gemma":[0.94480354,0.037169747,0.0019824733,0.005336419,0.009725976,0.0009818542],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016298663,0.0011941936,0.0008515729,0.003178012,0.00087721326,0.0029429372,0.0022212258,0.0014737472,0.005051651],"category_scores_gemma":[0.062275127,0.00089560065,0.00061087817,0.00093562517,0.0013034294,0.0045108567,0.0024358614,0.0014586006,0.002583568],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033949898,0.004124936,0.01822246,0.0019818065,0.0002151933,0.00078595185,0.013727141,0.011932718,0.06535889,0.041099474,0.007909065,0.8312473],"study_design_scores_gemma":[0.0020351713,0.006571565,0.017183717,0.0007274452,0.00048369367,0.0012650376,0.0037503063,0.6183553,0.22539376,0.043248624,0.08059192,0.00039354482],"about_ca_topic_score_codex":0.00045984847,"about_ca_topic_score_gemma":0.00040218377,"teacher_disagreement_score":0.016298663,"about_ca_system_score_codex":0.0010661829,"about_ca_system_score_gemma":0.0016214874,"threshold_uncertainty_score":0.0861966},"labels":[],"label_agreement":null},{"id":"W2973402705","doi":"10.5539/mas.v13n10p26","title":"Scaled Pearson’s Correlation Coefficient for Evaluating Text Similarity Measures","year":2019,"lang":"en","type":"article","venue":"Modern Applied Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Pearson product-moment correlation coefficient; Similarity (geometry); Correlation; Correlation coefficient; Outlier; Metric (unit); Statistics; Mathematics; Spearman's rank correlation coefficient; Computer science; Artificial intelligence","score_opus":0.03021596462251229,"score_gpt":0.31650101331620906,"score_spread":0.28628504869369675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2973402705","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21385749,0.008497422,0.7381066,0.00072812487,0.001050493,0.0014374874,0.013968815,0.0037702206,0.018583419],"genre_scores_gemma":[0.7126031,0.0014406252,0.27128327,0.00016266547,0.0002971312,0.0014881212,0.00985497,0.0004321593,0.0024379177],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97475344,0.00808569,0.003917849,0.0033345367,0.009454473,0.0004539459],"domain_scores_gemma":[0.92548674,0.047403727,0.0071806307,0.0054938584,0.013576645,0.00085837406],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011517358,0.0011760814,0.001282684,0.015147507,0.00082536566,0.0026082327,0.001184606,0.0012584524,0.00400707],"category_scores_gemma":[0.09467954,0.0002443036,0.0011310304,0.018563583,0.0010592676,0.0029155116,0.0013514445,0.001152027,0.0021932297],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083509757,0.00038618132,0.1910835,0.003238865,0.0022000968,0.00078414776,0.0020856136,0.03630426,0.013650253,0.026223991,0.03958389,0.68362415],"study_design_scores_gemma":[0.0001633869,0.0025478723,0.28798,0.0009871408,0.0008597286,0.002677573,0.004503324,0.50873476,0.022849618,0.068369776,0.09961668,0.00071020983],"about_ca_topic_score_codex":0.0021945229,"about_ca_topic_score_gemma":0.0029444515,"teacher_disagreement_score":0.015147507,"about_ca_system_score_codex":0.0009566001,"about_ca_system_score_gemma":0.0014059405,"threshold_uncertainty_score":0.060910344},"labels":[],"label_agreement":null},{"id":"W2973421725","doi":"10.1007/978-3-030-30712-7_46","title":"Visual Exploration of Topic Controversy in Online Conversations","year":2019,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Visualization; Data science; Online discussion; Work (physics); World Wide Web; Artificial intelligence; Engineering","score_opus":0.04436041703731572,"score_gpt":0.338240783602607,"score_spread":0.29388036656529126,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2973421725","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47116846,0.006073937,0.17166072,0.008158379,0.0010689598,0.00048799164,0.012690823,0.009540155,0.31915054],"genre_scores_gemma":[0.90365,0.0012767349,0.059699252,0.0006573841,0.00041401727,0.00025435374,0.0034476228,0.002110597,0.028490195],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998369,0.00006572046,0.000006048695,0.00002530033,0.000040774197,0.000025213256],"domain_scores_gemma":[0.9969014,0.0025241475,0.00014732344,0.00009006492,0.00017375624,0.0001633611],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004432085,0.00041072106,0.00021229831,0.0024215956,0.0009818415,0.0024696374,0.00049922924,0.00072789955,0.027438024],"category_scores_gemma":[0.0037653924,0.00020153516,0.00026793152,0.0019629034,0.00053820974,0.0026468642,0.001856716,0.0009681447,0.0020148444],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027779567,0.00020964755,0.01307856,0.0037287958,0.000102278434,0.0046117036,0.20049337,0.009697801,0.0680414,0.13947634,0.1625377,0.39524436],"study_design_scores_gemma":[0.00019841251,0.00021206703,0.043682463,0.0019776814,0.00012461274,0.002239511,0.080834895,0.10283366,0.01176959,0.13220371,0.62374365,0.00017978613],"about_ca_topic_score_codex":0.0024952674,"about_ca_topic_score_gemma":0.0038077089,"teacher_disagreement_score":0.027438024,"about_ca_system_score_codex":0.0005818609,"about_ca_system_score_gemma":0.00045213502,"threshold_uncertainty_score":0.091789305},"labels":[],"label_agreement":null},{"id":"W2974266994","doi":"10.1167/19.10.187","title":"Statistical learning enables implicit subadditive predictions","year":2019,"lang":"en","type":"article","venue":"Journal of Vision","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Subadditivity; Outcome (game theory); Object (grammar); Association (psychology); Psychology; Statistics; Statistical hypothesis testing; Coin flipping; Alternative hypothesis; Cognitive psychology; Mathematics; Artificial intelligence; Computer science; Combinatorics; Null hypothesis","score_opus":0.0061224596720312495,"score_gpt":0.2984555302704867,"score_spread":0.2923330705984555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2974266994","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7949628,0.0001792556,0.1913966,0.00071425224,0.00007528479,0.00011301888,0.00021919962,0.00075517665,0.011584374],"genre_scores_gemma":[0.9820227,0.00007097182,0.01685663,0.00010205421,0.00002246032,0.00003595168,0.00012046344,0.000043972992,0.0007247937],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99648726,0.0008908797,0.00020669914,0.0012626821,0.00090887625,0.00024359989],"domain_scores_gemma":[0.9660348,0.021346731,0.0044615753,0.0060917228,0.001259301,0.00080588117],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003698577,0.00076937873,0.0006257017,0.0004976101,0.0002596571,0.0015481013,0.0015223715,0.0009407086,0.0036750312],"category_scores_gemma":[0.04314071,0.00072070584,0.00067474396,0.00032371937,0.0015244053,0.0045111333,0.0023148695,0.0018125958,0.00054872816],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033270107,0.0013686663,0.10282084,0.0008890869,0.0004900424,0.0012580308,0.0023211995,0.07488552,0.22341575,0.08395271,0.0025231203,0.5027481],"study_design_scores_gemma":[0.00023428943,0.0017805767,0.07336619,0.000112599984,0.00023769362,0.0009041366,0.00030507747,0.4972409,0.08874352,0.33266726,0.00422161,0.00018623826],"about_ca_topic_score_codex":0.0012649298,"about_ca_topic_score_gemma":0.00095366983,"teacher_disagreement_score":0.003698577,"about_ca_system_score_codex":0.0007099342,"about_ca_system_score_gemma":0.00090209336,"threshold_uncertainty_score":0.019560158},"labels":[],"label_agreement":null},{"id":"W2974657091","doi":"10.4018/ijossp.2019070103","title":"A Topic Modeling Based Approach for Enhancing Corpus Querying","year":2019,"lang":"en","type":"article","venue":"International Journal of Open Source Software and Processes","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Query expansion; Ranking (information retrieval); Information retrieval; Query optimization; Process (computing); Selection (genetic algorithm); Web query classification; Web search query; Data mining; Query language; Sargable; Quality (philosophy); Search engine; Machine learning","score_opus":0.025355173247787816,"score_gpt":0.3171102658473063,"score_spread":0.2917550925995185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2974657091","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0073478287,0.0010010549,0.98497564,0.000336343,0.000099019315,0.00023066722,0.00043112235,0.0038903998,0.001687975],"genre_scores_gemma":[0.13534144,0.0013042642,0.85491544,0.00033096762,0.00034149128,0.00061425,0.0022832213,0.0009171915,0.0039517744],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966228,0.0012785769,0.00029893187,0.0007115548,0.00093100226,0.00015718343],"domain_scores_gemma":[0.9952657,0.0025606733,0.00021748384,0.00075779855,0.001076905,0.000121353834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036986824,0.001251312,0.0013129709,0.00391326,0.0012430506,0.0031695915,0.0018252981,0.0013943127,0.00249638],"category_scores_gemma":[0.011235467,0.00060229615,0.0018625071,0.005307352,0.00087392295,0.005439719,0.0026431545,0.0017898142,0.0017577359],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00074435864,0.00042391228,0.00390739,0.0010203439,0.00034957044,0.00061395305,0.0033454734,0.036728684,0.09143879,0.052723795,0.025637306,0.7830665],"study_design_scores_gemma":[0.000106362495,0.00033286863,0.0025176203,0.00008602324,0.0002793007,0.0010878714,0.0009231839,0.8379526,0.040357765,0.048755992,0.06739199,0.0002085062],"about_ca_topic_score_codex":0.0052189548,"about_ca_topic_score_gemma":0.005775075,"teacher_disagreement_score":0.0052189548,"about_ca_system_score_codex":0.0009262008,"about_ca_system_score_gemma":0.001551559,"threshold_uncertainty_score":0.019560695},"labels":[],"label_agreement":null},{"id":"W2982496367","doi":"","title":"Towards using task similarity to recommend Stack Overflow posts.","year":2018,"lang":"en","type":"article","venue":"Conferencia Iberoamericana de Software Engineering","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Task (project management); Stack (abstract data type); Similarity (geometry); Artificial intelligence; Programming language; Engineering","score_opus":0.020727173550247106,"score_gpt":0.27892915757006786,"score_spread":0.25820198401982075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2982496367","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.56632346,0.0085871685,0.38145298,0.002180879,0.0014889282,0.0012954101,0.010873951,0.011927625,0.015869593],"genre_scores_gemma":[0.8158738,0.0010684791,0.15561678,0.0004365964,0.0010692507,0.00042322304,0.01291056,0.00042620493,0.012175197],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99767905,0.0006447408,0.00024175039,0.00057142717,0.0006177794,0.000245244],"domain_scores_gemma":[0.9926387,0.00372756,0.000719458,0.00050949043,0.0018501218,0.0005546566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020480666,0.0010694147,0.0011988762,0.008763984,0.0010969124,0.002357328,0.0010982973,0.0019042278,0.002988066],"category_scores_gemma":[0.0142065175,0.00031170636,0.0007553516,0.0035257898,0.00034640136,0.0039098547,0.0011532552,0.0013806289,0.0037078327],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026519112,0.0018024391,0.04521539,0.0009721622,0.00046127327,0.0003470585,0.0009709373,0.011311558,0.030716676,0.0031031023,0.06665919,0.83578837],"study_design_scores_gemma":[0.00028123302,0.0011866963,0.039563697,0.00019251285,0.0004392804,0.0005274533,0.001910793,0.8893216,0.024240758,0.015803799,0.026371388,0.00016074424],"about_ca_topic_score_codex":0.007708355,"about_ca_topic_score_gemma":0.0136685185,"teacher_disagreement_score":0.008763984,"about_ca_system_score_codex":0.0007930664,"about_ca_system_score_gemma":0.0021135733,"threshold_uncertainty_score":0.015326917},"labels":[],"label_agreement":null},{"id":"W2990795924","doi":"10.1177/2515245919882693","title":"Advancing Meta-Analysis With Knowledge-Management Platforms: Using metaBUS in Psychology","year":2019,"lang":"en","type":"article","venue":"Advances in Methods and Practices in Psychological Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Northern Alberta Institute of Technology","funders":"","keywords":"Knowledge management; Data science; Computer science; Numbering; Visualization; Engineering ethics; World Wide Web; Psychology; Engineering","score_opus":0.11064732021659938,"score_gpt":0.5827248165838096,"score_spread":0.4720774963672102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2990795924","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0051778746,0.10230505,0.81419337,0.028486468,0.0025678482,0.0023057018,0.010004819,0.024112122,0.010846751],"genre_scores_gemma":[0.021685451,0.024784764,0.93721884,0.002619612,0.0010075074,0.0042889444,0.0041092196,0.0033328733,0.0009528074],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.8297907,0.12674049,0.020099422,0.0071553634,0.015604566,0.00060956326],"domain_scores_gemma":[0.2452169,0.66281015,0.019161265,0.053599533,0.016745377,0.0024668232],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2446642,0.0033285916,0.005121017,0.03710539,0.0028800373,0.017891947,0.0051018656,0.0030856554,0.013336965],"category_scores_gemma":[0.55015075,0.002691045,0.007039639,0.02841002,0.004139376,0.017039519,0.014943393,0.006716573,0.0048807086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061866094,0.0001732076,0.005326139,0.07780028,0.009401785,0.0006769831,0.013106609,0.004927452,0.0034732078,0.10595024,0.06846631,0.71007913],"study_design_scores_gemma":[0.00038493564,0.000244159,0.0052884985,0.05703068,0.005431303,0.00065951544,0.0023393943,0.015576757,0.0053219516,0.40432915,0.5028402,0.0005534642],"about_ca_topic_score_codex":0.0021980153,"about_ca_topic_score_gemma":0.004851283,"teacher_disagreement_score":0.7553358,"about_ca_system_score_codex":0.003962976,"about_ca_system_score_gemma":0.016815491,"threshold_uncertainty_score":0.9314635},"labels":[],"label_agreement":null},{"id":"W2994813639","doi":"10.5539/cis.v13n3p57","title":"Topic Subject Creation Using Unsupervised Learning for Topic Modeling","year":2020,"lang":"en","type":"preprint","venue":"Computer and Information Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Latent Dirichlet allocation; Topic model; Non-negative matrix factorization; Subject (documents); Computer science; Artificial intelligence; Machine learning; Matrix decomposition; Data science; World Wide Web","score_opus":0.050476913095002,"score_gpt":0.32253521218980263,"score_spread":0.2720582990948006,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2994813639","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011882168,0.00043793398,0.9856942,0.00015574905,0.000078718695,0.00011374509,0.0002292673,0.00081875804,0.00058950187],"genre_scores_gemma":[0.22264421,0.0008025294,0.76883453,0.00017370786,0.00062478083,0.0006712835,0.0027414656,0.00034779776,0.0031596487],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972146,0.0013694585,0.00013135755,0.0007446193,0.00039181052,0.00014814221],"domain_scores_gemma":[0.98956674,0.007851208,0.000541093,0.0010889474,0.0007399925,0.00021196104],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043394286,0.000983068,0.0012131477,0.0031829996,0.00094345695,0.001837953,0.0015966508,0.0010984532,0.0014923221],"category_scores_gemma":[0.012317847,0.00043220836,0.0017096805,0.0030723203,0.00086721097,0.0023958737,0.0014135003,0.0018558239,0.0016304726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005142719,0.0007870633,0.008453275,0.000631542,0.00038912997,0.0002664395,0.002205728,0.094787724,0.018762268,0.026335629,0.011067995,0.83579886],"study_design_scores_gemma":[0.00003313408,0.00012894381,0.0018422885,0.000030953302,0.00006914844,0.0001669456,0.00014902484,0.9560905,0.006688599,0.027298355,0.0074529196,0.0000492262],"about_ca_topic_score_codex":0.002016339,"about_ca_topic_score_gemma":0.0029318822,"teacher_disagreement_score":0.0043394286,"about_ca_system_score_codex":0.0005823014,"about_ca_system_score_gemma":0.0011238937,"threshold_uncertainty_score":0.022949338},"labels":[],"label_agreement":null},{"id":"W3004295495","doi":"10.22148/001c.11829","title":"On the perceived complexity of literature. A response to Nan Z. Da","year":2020,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Reductionism; Variance (accounting); Literary criticism; Epistemology; Sociology; Linguistics; Philosophy; Economics","score_opus":0.07676143944045041,"score_gpt":0.3253034171408474,"score_spread":0.24854197770039696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004295495","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00039401205,0.0038003686,0.00013301504,0.98994243,0.004919623,0.0000037619916,0.000023775476,0.000010337414,0.0007725911],"genre_scores_gemma":[0.012097589,0.004577415,0.0005123787,0.9684254,0.010656457,0.000042033727,0.000030179706,0.0000686082,0.0035900797],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9876373,0.0052866978,0.00095868914,0.0020411264,0.0032801663,0.0007959881],"domain_scores_gemma":[0.89164627,0.08153684,0.004357274,0.0027577677,0.0135303615,0.006171548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013765807,0.0013216302,0.0019770232,0.0021160054,0.010687716,0.014971939,0.0032981606,0.025539372,0.0066182264],"category_scores_gemma":[0.06551142,0.000915264,0.000903907,0.0029089167,0.019528002,0.023793703,0.011362213,0.050710306,0.0025093108],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003167514,0.000015871947,0.00038803645,0.00013450612,0.000015656306,0.00019134463,0.007828004,0.00004157794,0.0001324799,0.017442452,0.96646,0.007318384],"study_design_scores_gemma":[0.000024278295,0.000022887421,0.0011332775,0.00081099366,0.000011888965,0.00068435684,0.01881252,0.00016200935,0.00017918885,0.02143895,0.9566278,0.000091772505],"about_ca_topic_score_codex":0.009142978,"about_ca_topic_score_gemma":0.0118873045,"teacher_disagreement_score":0.025539372,"about_ca_system_score_codex":0.006815816,"about_ca_system_score_gemma":0.0061076996,"threshold_uncertainty_score":0.07280147},"labels":[],"label_agreement":null},{"id":"W3007728180","doi":"10.1109/icmla.2019.00228","title":"Using Convolutional Neural Networks to Extract Keywords and Keyphrases: A Case Study for Foodborne Illnesses","year":2019,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Computer science; Convolutional neural network; Artificial intelligence; Benchmark (surveying); Identification (biology); Generalization; Field (mathematics); Deep learning; Information extraction; Natural language processing; Machine learning; Information retrieval","score_opus":0.040456376868748964,"score_gpt":0.344727807278998,"score_spread":0.304271430410249,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3007728180","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97912955,0.0007652002,0.011252242,0.0010830406,0.00005527651,0.00017501372,0.0029195116,0.00033168847,0.004288403],"genre_scores_gemma":[0.951566,0.000779768,0.037072003,0.0002487845,0.000047190813,0.00008853579,0.0035809872,0.000058168825,0.0065585417],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995252,0.00012670318,0.000047977494,0.00007755676,0.00015067265,0.000071844246],"domain_scores_gemma":[0.9982974,0.0010835148,0.00012682627,0.00012519206,0.00029023705,0.000076765056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00062611484,0.000549181,0.00028225814,0.0010996169,0.00054693996,0.000470802,0.0006399116,0.001009585,0.00091191364],"category_scores_gemma":[0.0025158967,0.00015812185,0.00043979703,0.0011258074,0.00043076114,0.0008286538,0.00037264713,0.00043655178,0.00032579128],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002010761,0.0025013553,0.23674396,0.0031952555,0.00038066215,0.06680035,0.00529284,0.08084874,0.061614383,0.0042999117,0.032958273,0.50335354],"study_design_scores_gemma":[0.00023590995,0.0016571354,0.24239703,0.0003653597,0.00038371838,0.022993613,0.010457936,0.4858649,0.16567466,0.0050847447,0.06466274,0.00022228406],"about_ca_topic_score_codex":0.02363242,"about_ca_topic_score_gemma":0.058150385,"teacher_disagreement_score":0.02363242,"about_ca_system_score_codex":0.0011981569,"about_ca_system_score_gemma":0.0006638273,"threshold_uncertainty_score":0.04698968},"labels":[],"label_agreement":null},{"id":"W3014604514","doi":"10.2196/17642","title":"Using Natural Language Processing Techniques to Provide Personalized Educational Materials for Chronic Disease Patients in China: Development and Assessment of a Knowledge-Based Health Recommender System","year":2020,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Ontology; Computer science; Recommender system; Information retrieval; Artificial intelligence","score_opus":0.027770495797533334,"score_gpt":0.38999263871658607,"score_spread":0.36222214291905275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3014604514","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9242926,0.00057463354,0.06853702,0.0007755049,0.00005007456,0.0009334297,0.00064831803,0.001812951,0.002375455],"genre_scores_gemma":[0.78136116,0.00065967557,0.21260281,0.0002620912,0.000025155032,0.0004161712,0.0017678852,0.000036185953,0.0028689634],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918336,0.00021563943,0.00012688665,0.00020886443,0.000206325,0.000058940845],"domain_scores_gemma":[0.99803954,0.00087365875,0.00012829361,0.000158864,0.00071613776,0.00008360877],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002209138,0.00048224165,0.0005124989,0.0009825816,0.00053563266,0.00067154627,0.00081813545,0.0006373953,0.00090275414],"category_scores_gemma":[0.004017827,0.00019970475,0.00058606506,0.0006209242,0.00021523592,0.0010394047,0.00047876503,0.00033847633,0.00029156735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007198158,0.0019753382,0.08721629,0.0009941171,0.0002947433,0.0011983842,0.0017539248,0.028867677,0.049754996,0.0011293307,0.005388393,0.8207071],"study_design_scores_gemma":[0.00034256015,0.0021848364,0.09433893,0.00012967271,0.00085227774,0.00084687845,0.0017324357,0.8454027,0.043220177,0.00086626527,0.009923846,0.00015945135],"about_ca_topic_score_codex":0.032431796,"about_ca_topic_score_gemma":0.02859724,"teacher_disagreement_score":0.032431796,"about_ca_system_score_codex":0.0011400458,"about_ca_system_score_gemma":0.0021428913,"threshold_uncertainty_score":0.06448603},"labels":[],"label_agreement":null},{"id":"W3015234797","doi":"10.36227/techrxiv.12100692.v1","title":"Deep Learning for text in limted data settings","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Computer science; Artificial intelligence; Deep learning; Sequence (biology); Transfer of learning; Sequence learning; Natural language processing; Machine learning; Recurrent neural network; Sentiment analysis; Artificial neural network","score_opus":0.05509319530841789,"score_gpt":0.34010155808589826,"score_spread":0.2850083627774804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3015234797","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010233572,0.0027297619,0.9730117,0.003458356,0.00031498403,0.00012590937,0.0022978394,0.0028328446,0.00499507],"genre_scores_gemma":[0.41089734,0.006216614,0.5476781,0.0014419495,0.0009396673,0.00077860453,0.009045492,0.0008122268,0.022189943],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988223,0.00044705824,0.00008738157,0.00029332945,0.00023858207,0.00011128774],"domain_scores_gemma":[0.9972805,0.0016485598,0.0001699138,0.00044433327,0.00036516308,0.000091536356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002402293,0.0011391226,0.0008336513,0.0012132971,0.00057915197,0.0021047397,0.0018806083,0.0023597672,0.011237511],"category_scores_gemma":[0.010067598,0.00061264104,0.0007771679,0.0016427924,0.0009104633,0.0053395573,0.0024121017,0.00391023,0.0044866763],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003561153,0.00031393694,0.0024259884,0.0009290323,0.00018964634,0.00051352294,0.00023171758,0.27146766,0.004538766,0.123874396,0.05348805,0.5416712],"study_design_scores_gemma":[0.000014339488,0.000030067878,0.00030752123,0.000044532473,0.000008001712,0.000046655303,0.000028728291,0.8908821,0.0011521878,0.099991515,0.0074841096,0.00001032168],"about_ca_topic_score_codex":0.0052685784,"about_ca_topic_score_gemma":0.009025916,"teacher_disagreement_score":0.011237511,"about_ca_system_score_codex":0.0016928843,"about_ca_system_score_gemma":0.0010481182,"threshold_uncertainty_score":0.037593186},"labels":[],"label_agreement":null},{"id":"W3023453969","doi":"10.1007/978-3-030-47358-7_19","title":"Using Topic Modelling to Improve Prediction of Financial Report Commentary Classes","year":2020,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Latent Dirichlet allocation; Class (philosophy); Task (project management); Feature selection; Selection (genetic algorithm); Artificial intelligence; Machine learning; Data mining; Feature (linguistics); Finance; Topic model","score_opus":0.03398848257873614,"score_gpt":0.28170749902452086,"score_spread":0.2477190164457847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3023453969","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7660259,0.010581187,0.16506119,0.0036291338,0.002310241,0.00034840725,0.026867004,0.010244605,0.014932288],"genre_scores_gemma":[0.9295074,0.0012588039,0.03148069,0.00023515736,0.0015674116,0.00021811394,0.027235478,0.0003321273,0.00816484],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986266,0.0005757197,0.000111898036,0.00033218824,0.00021873393,0.00013495317],"domain_scores_gemma":[0.9831638,0.013511438,0.00090427283,0.0004617838,0.0014989439,0.00045987693],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033263983,0.0011337817,0.0009005132,0.004098197,0.00049949146,0.0022246821,0.000973158,0.0014608226,0.0037598407],"category_scores_gemma":[0.0148873655,0.00036768068,0.0014806272,0.0027364227,0.0001742633,0.0017793424,0.0007264304,0.0021529652,0.0045057978],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037010887,0.0017180484,0.15788893,0.000842862,0.0007564126,0.0003115476,0.000656028,0.0757333,0.012031632,0.0019741491,0.08261554,0.66177046],"study_design_scores_gemma":[0.00010827653,0.00033533486,0.050793972,0.00008879005,0.00034872317,0.00010749787,0.00027152034,0.930118,0.005760497,0.0032033343,0.008803004,0.00006107242],"about_ca_topic_score_codex":0.010677435,"about_ca_topic_score_gemma":0.010432324,"teacher_disagreement_score":0.010677435,"about_ca_system_score_codex":0.0008596048,"about_ca_system_score_gemma":0.0007776816,"threshold_uncertainty_score":0.021230578},"labels":[],"label_agreement":null},{"id":"W3023942231","doi":"10.1075/ml.20004.nis","title":"Clozapp","year":2019,"lang":"en","type":"article","venue":"The Mental Lexicon","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Cégep André Laurendeau","funders":"","keywords":"Computer science; Data collection; Replication (statistics); Predictability; Information retrieval; Java; Natural language processing; Artificial intelligence; Programming language; Statistics","score_opus":0.007162570725038875,"score_gpt":0.2528083073504823,"score_spread":0.24564573662544345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3023942231","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027159486,0.0035656986,0.30123335,0.0017863526,0.000994883,0.0077134566,0.05006565,0.40988943,0.19759168],"genre_scores_gemma":[0.13587376,0.002633952,0.49343672,0.0035617119,0.00044984015,0.012917491,0.054720886,0.07130907,0.22509658],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.996563,0.0007972309,0.00039120927,0.00074124505,0.0013030452,0.00020429098],"domain_scores_gemma":[0.97975355,0.009674103,0.0010027857,0.0037463543,0.0052356618,0.0005873986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029397935,0.0017993913,0.0012645754,0.004303583,0.0010736325,0.0034427599,0.0022861995,0.0013250962,0.12115579],"category_scores_gemma":[0.021987543,0.0011427298,0.0008656308,0.0025806648,0.00072373933,0.003430494,0.0035987487,0.0014500155,0.063315846],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001659936,0.00032517753,0.0032716224,0.0030429333,0.00007849344,0.0008826008,0.0017157228,0.00040799333,0.020757783,0.007045128,0.31780314,0.6430094],"study_design_scores_gemma":[0.0003293362,0.0004120943,0.011474832,0.001178572,0.00009774418,0.0021335066,0.00061346794,0.0042623766,0.026280835,0.014082978,0.9388429,0.00029145254],"about_ca_topic_score_codex":0.0027553733,"about_ca_topic_score_gemma":0.0051914738,"teacher_disagreement_score":0.12115579,"about_ca_system_score_codex":0.0005944986,"about_ca_system_score_gemma":0.001912724,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W3027864066","doi":"10.1145/3377939","title":"Story Forest","year":2020,"lang":"en","type":"article","venue":"ACM Transactions on Knowledge Discovery from Data","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Timeline; Computer science; Event (particle physics); Information retrieval; The Internet; Novelty; World Wide Web; Cluster analysis; Set (abstract data type); News aggregator; Graph; Data science; Artificial intelligence; History","score_opus":0.06664237805374586,"score_gpt":0.31143453917885444,"score_spread":0.2447921611251086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3027864066","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01626917,0.0039834026,0.5543593,0.002554548,0.0015691004,0.0021011687,0.1660955,0.089188926,0.16387884],"genre_scores_gemma":[0.09396937,0.002320024,0.48769617,0.0013040793,0.00039541433,0.0013803239,0.33396837,0.005827719,0.07313858],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99916327,0.00012693045,0.000059221944,0.00031697543,0.00023451386,0.00009914594],"domain_scores_gemma":[0.9990376,0.00032152433,0.000065522654,0.00025934642,0.00023483818,0.0000811186],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006922385,0.0017553822,0.0006985309,0.0031074923,0.0013317042,0.0026058871,0.0022977763,0.0013025184,0.07456609],"category_scores_gemma":[0.0038586634,0.00062553026,0.0018523503,0.0025705355,0.0003868566,0.004175462,0.0022834626,0.0014784132,0.03810922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035075351,0.0001809252,0.0028700111,0.00090498273,0.00013052174,0.00045538624,0.00024213098,0.012356058,0.002915271,0.02594774,0.4491531,0.50449306],"study_design_scores_gemma":[0.00013533577,0.000116324656,0.00206778,0.00023571418,0.00010193977,0.0010177935,0.00044214897,0.17169416,0.0063846027,0.08241154,0.73532397,0.000068710135],"about_ca_topic_score_codex":0.004977882,"about_ca_topic_score_gemma":0.0142069375,"teacher_disagreement_score":0.07456609,"about_ca_system_score_codex":0.00067785446,"about_ca_system_score_gemma":0.0013831878,"threshold_uncertainty_score":0.24944842},"labels":[],"label_agreement":null},{"id":"W3031557112","doi":"10.18653/v1/2020.coling-main.6","title":"Catching Attention with Automatic Pull Quote Selection","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Automatic summarization; Task (project management); Readability; Computer science; Selection (genetic algorithm); Identification (biology); Presentation (obstetrics); Salient; Code (set theory); Natural language processing; Artificial intelligence; Perception; Cognitive psychology; Psychology","score_opus":0.01647314170779892,"score_gpt":0.2747997570315964,"score_spread":0.2583266153237975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3031557112","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22463945,0.0015321538,0.74041957,0.0012561581,0.00044184917,0.00045989585,0.0007586914,0.016787753,0.013704458],"genre_scores_gemma":[0.820718,0.00033703865,0.16236274,0.00038818226,0.0003035818,0.00033018924,0.0012380391,0.00082011736,0.013502076],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9974425,0.0010929465,0.00011783774,0.000623095,0.0005327737,0.00019082606],"domain_scores_gemma":[0.9915719,0.0049865306,0.00077531143,0.0010108912,0.0011963256,0.0004590148],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025117921,0.0011892698,0.00088193465,0.0012072299,0.00053193385,0.002292209,0.0011641308,0.0014822001,0.006869742],"category_scores_gemma":[0.017018486,0.00039839972,0.0006428653,0.000707036,0.00055217696,0.004322073,0.002615645,0.0014804037,0.0056398567],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021691108,0.0005867613,0.010695306,0.0010460495,0.00015700815,0.0005901593,0.0031941065,0.011321505,0.14457123,0.006462157,0.017621333,0.80158526],"study_design_scores_gemma":[0.00026038618,0.0018007815,0.02780293,0.00027543207,0.00018742324,0.001233002,0.002783156,0.7148579,0.14259519,0.06570753,0.04223468,0.0002616022],"about_ca_topic_score_codex":0.0006197959,"about_ca_topic_score_gemma":0.0011688643,"teacher_disagreement_score":0.006869742,"about_ca_system_score_codex":0.0003745285,"about_ca_system_score_gemma":0.00054435025,"threshold_uncertainty_score":0.022981524},"labels":[],"label_agreement":null},{"id":"W3034847701","doi":"10.5430/ijhe.v9n4p169","title":"Analysis of Text Mining from Full-text Articles and Abstracts by Postgraduates Students in Selected Nigeria Universities","year":2020,"lang":"en","type":"article","venue":"International Journal of Higher Education","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Originality; Computer science; Data collection; Library science; World Wide Web; Psychology; Mathematics education; Medical education; Information retrieval; Medicine; Sociology; Qualitative research; Social science","score_opus":0.010294624633854506,"score_gpt":0.307760733735449,"score_spread":0.2974661091015945,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3034847701","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9588948,0.0070145866,0.0076713944,0.0036566637,0.00043648886,0.0035209819,0.011103347,0.0002307649,0.007471159],"genre_scores_gemma":[0.9142117,0.012689641,0.04368389,0.001629951,0.0008487329,0.006199371,0.011887716,0.00017959581,0.008669418],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99332994,0.0020836538,0.0017725986,0.000603666,0.0019284879,0.000281643],"domain_scores_gemma":[0.9167146,0.050840117,0.0122041,0.0018126625,0.016572086,0.0018564212],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007634984,0.0003821008,0.0008069529,0.008811904,0.00092207344,0.0022710725,0.00075627246,0.00056362053,0.0033804805],"category_scores_gemma":[0.05659545,0.00021233676,0.0005328107,0.0076811677,0.0005588018,0.002004243,0.0011804546,0.0005601499,0.0012901734],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013039618,0.0007319685,0.13909851,0.02428494,0.00028098785,0.004696459,0.06584944,0.00044872786,0.018870922,0.00096633204,0.029068904,0.7143988],"study_design_scores_gemma":[0.00015115837,0.0021435893,0.5934014,0.008709858,0.0006062363,0.0050415997,0.18606149,0.0022643125,0.020606054,0.0028955848,0.1778923,0.00022648233],"about_ca_topic_score_codex":0.0006991219,"about_ca_topic_score_gemma":0.001522631,"teacher_disagreement_score":0.008811904,"about_ca_system_score_codex":0.00077705755,"about_ca_system_score_gemma":0.0031979687,"threshold_uncertainty_score":0.040378153},"labels":[],"label_agreement":null},{"id":"W3035454674","doi":"","title":"Meta Variance Transfer: Learning to Augment from the Others","year":2020,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Augment; Computer science; Variance (accounting); Artificial intelligence","score_opus":0.044351630801325675,"score_gpt":0.26800000279086234,"score_spread":0.22364837198953666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035454674","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030776778,0.001027795,0.9583407,0.0006621485,0.00041193172,0.000102538004,0.0002682401,0.004754788,0.0036551398],"genre_scores_gemma":[0.6477566,0.0007540584,0.3382169,0.00072894496,0.0008549908,0.00033923276,0.0011700686,0.0010452355,0.009134042],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99710983,0.0012687325,0.00014399365,0.0008419896,0.00047409115,0.00016152137],"domain_scores_gemma":[0.9924069,0.0043970174,0.00037181782,0.0017202226,0.00091754214,0.00018645337],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004854645,0.0018517678,0.0013238784,0.001902408,0.00088117155,0.0018949701,0.0022718345,0.002042233,0.0046937875],"category_scores_gemma":[0.018058484,0.00057368085,0.0017943864,0.0017521316,0.0010233627,0.004617913,0.0038661999,0.003038664,0.0027508445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007291403,0.0006356501,0.003466812,0.00024165206,0.00050208275,0.0001078583,0.0004006623,0.036225583,0.014922579,0.0116142435,0.013356811,0.91779685],"study_design_scores_gemma":[0.000094205665,0.00048452406,0.0017379792,0.00008489121,0.0003386364,0.00013675194,0.00013730497,0.9165678,0.015336999,0.05999345,0.005021064,0.000066484674],"about_ca_topic_score_codex":0.000998382,"about_ca_topic_score_gemma":0.0014352158,"teacher_disagreement_score":0.004854645,"about_ca_system_score_codex":0.00056041,"about_ca_system_score_gemma":0.0011493865,"threshold_uncertainty_score":0.025674164},"labels":[],"label_agreement":null},{"id":"W3038540624","doi":"10.1007/978-3-030-43961-3_3","title":"Animated Guide to Represent a Novel Means of Gut-Brain Axis Communication","year":2020,"lang":"en","type":"article","venue":"Advances in experimental medicine and biology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institute of Infection and Immunity","funders":"","keywords":"Neuroscience; Computer science; Cognitive science; Biology; Psychology","score_opus":0.04733392472226366,"score_gpt":0.4147131304726079,"score_spread":0.3673792057503442,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3038540624","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029930107,0.0007264389,0.8434442,0.0021658875,0.0036226192,0.00046023895,0.009417387,0.033555914,0.10361424],"genre_scores_gemma":[0.029240415,0.0010325463,0.78914243,0.0017221295,0.00041720032,0.0010854314,0.0053568743,0.00816529,0.1638376],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99988306,0.000031978085,0.0000068930353,0.000025097466,0.000035768357,0.000017203649],"domain_scores_gemma":[0.99954,0.00022683723,0.00001694737,0.0000561135,0.000107818756,0.000052366166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00023809317,0.0010418389,0.00028853878,0.00067245757,0.00048992154,0.0013436083,0.0012189716,0.0014270602,0.17931943],"category_scores_gemma":[0.0017858554,0.00025809597,0.00050138356,0.00040216095,0.00064209994,0.0009031068,0.0013914687,0.0011073274,0.053663705],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005356915,0.000117866526,0.00036196716,0.0007469355,0.000031059302,0.00079109543,0.000761597,0.0073094903,0.04044627,0.0863135,0.61617726,0.2464073],"study_design_scores_gemma":[0.00005551849,0.00005676216,0.00026212036,0.00013548452,0.000013389324,0.00034681673,0.000098209595,0.020848792,0.009873682,0.0143693965,0.95390946,0.000030441965],"about_ca_topic_score_codex":0.0023691244,"about_ca_topic_score_gemma":0.00434789,"teacher_disagreement_score":0.17931943,"about_ca_system_score_codex":0.00032509537,"about_ca_system_score_gemma":0.0006190013,"threshold_uncertainty_score":0.5998832},"labels":[],"label_agreement":null},{"id":"W3042358492","doi":"10.17705/1jais.00627","title":"Basic Classes in Conceptual Modeling: Theory and Practical Guidelines","year":2020,"lang":"en","type":"article","venue":"Journal of the Association for Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"HEC Montréal","funders":"","keywords":"Computer science; Variety (cybernetics); Domain (mathematical analysis); Conceptual model; Conceptual framework; Interface (matter); Diversity (politics); Management science; Data science; Information system; Knowledge management; Artificial intelligence; Epistemology; Mathematics; Engineering; Database; Sociology","score_opus":0.06347917459344107,"score_gpt":0.34107894678391293,"score_spread":0.2775997721904718,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3042358492","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034129908,0.0052910843,0.94186527,0.021708613,0.00029527623,0.0008750882,0.00026775975,0.00046071576,0.025823131],"genre_scores_gemma":[0.074976034,0.003801788,0.9143173,0.0017466124,0.00040344064,0.0021149323,0.00043830252,0.00015381977,0.002047772],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97947663,0.013573549,0.0021569803,0.0017531394,0.0025322211,0.00050755555],"domain_scores_gemma":[0.9630693,0.026369933,0.0016564667,0.0040792413,0.0034507404,0.0013744003],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03608807,0.0020622008,0.0024335391,0.009183049,0.0057106568,0.015227176,0.007887491,0.0067133005,0.0067690234],"category_scores_gemma":[0.044350766,0.0025683444,0.0027506773,0.0084890025,0.034826197,0.03351527,0.0074473997,0.011695214,0.0025306505],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000049973564,0.000017991744,0.0001386294,0.00011176289,0.0000063667676,0.000029438428,0.0014376742,0.0004933591,0.00004018809,0.9889538,0.0017656042,0.007000113],"study_design_scores_gemma":[0.000009991873,0.0000056837157,0.00006356811,0.0002002933,0.000004521235,0.000047138197,0.00058150623,0.0024940516,0.000051148072,0.97900087,0.017529674,0.000011427042],"about_ca_topic_score_codex":0.008541661,"about_ca_topic_score_gemma":0.0067784395,"teacher_disagreement_score":0.03608807,"about_ca_system_score_codex":0.008885882,"about_ca_system_score_gemma":0.007156376,"threshold_uncertainty_score":0.19085419},"labels":[],"label_agreement":null},{"id":"W3043122048","doi":"10.56042/alis.v67i1.28307","title":"Citations in chemical engineering research: factors and their assessment","year":2020,"lang":"en","type":"article","venue":"Annals of Library and Information Studies","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Citation; China; Impact factor; Library science; Web of science; Geography; Political science; Computer science; MEDLINE; Archaeology; Law","score_opus":0.20239466047734267,"score_gpt":0.3965406839996783,"score_spread":0.19414602352233562,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3043122048","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9547552,0.01328268,0.0051482725,0.001607783,0.00015948674,0.0004967061,0.001908548,0.00014353197,0.022497859],"genre_scores_gemma":[0.9956312,0.0011747248,0.0017389841,0.000036913185,0.000084575826,0.00007633185,0.00041464396,0.000015155919,0.00082744134],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9717077,0.005510327,0.0051828898,0.0012404107,0.015447251,0.0009113603],"domain_scores_gemma":[0.8020931,0.11843789,0.0390658,0.0034278503,0.03307083,0.0039044356],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.01597304,0.00045537102,0.0008845748,0.035000466,0.0013421413,0.005612742,0.0009324663,0.0012367412,0.0027851625],"category_scores_gemma":[0.1377233,0.00021920825,0.0014699306,0.05774207,0.0017232603,0.0040793624,0.0021492478,0.0008863321,0.00094501243],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000078186466,0.00006382939,0.95923424,0.0003068311,0.00021208213,0.00013491268,0.0012032891,0.00030091827,0.0001423742,0.0010355624,0.00070046436,0.036587197],"study_design_scores_gemma":[0.00001224815,0.00022026295,0.9853674,0.0002494576,0.00023419851,0.0006124797,0.0030133883,0.0019722444,0.00037436883,0.0022236395,0.0056664213,0.000053835203],"about_ca_topic_score_codex":0.0038244256,"about_ca_topic_score_gemma":0.0034392788,"teacher_disagreement_score":0.98402697,"about_ca_system_score_codex":0.0027654555,"about_ca_system_score_gemma":0.0030627325,"threshold_uncertainty_score":0.08447456},"labels":[],"label_agreement":null},{"id":"W3080743972","doi":"10.11606/d.55.2020.tde-20082020-093906","title":"Interactive keyterm-based document clustering and visualization via neural language models","year":2020,"lang":"en","type":"dissertation","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico; Dalhousie University; Indiana Corn Marketing Council","keywords":"Cluster analysis; Visualization; Computer science; Natural language processing; Artificial neural network; Artificial intelligence","score_opus":0.009807791098549257,"score_gpt":0.3188874140990868,"score_spread":0.30907962300053754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3080743972","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015358967,0.0012200219,0.89988005,0.0010712771,0.00048664518,0.00014176566,0.005605251,0.0697214,0.0065146224],"genre_scores_gemma":[0.14057341,0.0012838145,0.82629925,0.00020454104,0.00021527022,0.00024406993,0.0066818274,0.005326389,0.019171499],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956316,0.00010221609,0.00003197209,0.00011945474,0.00014806814,0.000035078858],"domain_scores_gemma":[0.99867,0.0006460105,0.00005679855,0.00018916429,0.00034153153,0.00009653424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006936283,0.0009462635,0.0005920544,0.0019127923,0.0004735899,0.0026688233,0.0009191135,0.00075184065,0.031718876],"category_scores_gemma":[0.0041642785,0.00035059682,0.00082190265,0.0017486871,0.00022791528,0.002607403,0.0013735442,0.0012481981,0.01148756],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069993583,0.00017498212,0.00094239047,0.0006143027,0.000120078264,0.00018809167,0.00045894203,0.018277507,0.056384996,0.006794925,0.08657652,0.8287673],"study_design_scores_gemma":[0.00012285831,0.00012706446,0.0019513137,0.000121678175,0.00007646356,0.0002964693,0.0003613494,0.8238013,0.082559735,0.023137545,0.067339845,0.00010434261],"about_ca_topic_score_codex":0.004452305,"about_ca_topic_score_gemma":0.008483931,"teacher_disagreement_score":0.031718876,"about_ca_system_score_codex":0.0005171529,"about_ca_system_score_gemma":0.0006853558,"threshold_uncertainty_score":0.106110215},"labels":[],"label_agreement":null},{"id":"W3082570699","doi":"10.48550/arxiv.2106.14731","title":"The DELICES project: Indexing scientific literature through semantic expansion","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Search engine indexing; Computer science; Relevance (law); Information retrieval; Scientific literature; Digital library; Data science; World Wide Web; Linguistics; Political science","score_opus":0.06650420550289016,"score_gpt":0.2294048143742566,"score_spread":0.16290060887136645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3082570699","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049016036,0.022499926,0.7677624,0.008588606,0.0032646847,0.0026297313,0.05435422,0.034663744,0.057220615],"genre_scores_gemma":[0.08412324,0.009002444,0.8019573,0.0012534101,0.0013904332,0.001336167,0.0791132,0.0027133133,0.019110536],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99391514,0.0023694707,0.0005974687,0.0009925633,0.0018767924,0.0002485125],"domain_scores_gemma":[0.9916788,0.00363503,0.0007501487,0.0017421364,0.0017514521,0.00044227848],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0054478953,0.0014969619,0.0013387084,0.017252969,0.0013468196,0.0034758581,0.0013045141,0.0011293886,0.0075356807],"category_scores_gemma":[0.016306669,0.00049742346,0.0015332745,0.012415515,0.0020116847,0.009864902,0.0059029656,0.002017339,0.0061928346],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007661233,0.0004137192,0.0016676072,0.0031845486,0.0002522285,0.00034683573,0.0015884582,0.0036818678,0.024020819,0.09960707,0.13209683,0.73237395],"study_design_scores_gemma":[0.0005493388,0.0004611519,0.0036537554,0.0009610966,0.00034396682,0.00070377387,0.0014419942,0.0617405,0.031740084,0.21569137,0.6825021,0.00021088643],"about_ca_topic_score_codex":0.002799188,"about_ca_topic_score_gemma":0.0032114843,"teacher_disagreement_score":0.99652416,"about_ca_system_score_codex":0.0013163346,"about_ca_system_score_gemma":0.0035316178,"threshold_uncertainty_score":0.028811574},"labels":[],"label_agreement":null},{"id":"W3087953246","doi":"","title":"Strategic Categorization, Category Bundle, and Typecasting: Three Essays on Product Categorization","year":2020,"lang":"en","type":"dissertation","venue":"York University Digital Library (York University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"University of Southern California","keywords":"Categorization; Product (mathematics); Product category; Bundle; Computer science; Artificial intelligence; Mathematics; Materials science","score_opus":0.01715802069653554,"score_gpt":0.18401840740315073,"score_spread":0.16686038670661518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3087953246","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.096867755,0.023276396,0.12720524,0.13860507,0.014158605,0.0006168636,0.00022408462,0.00022732644,0.5988187],"genre_scores_gemma":[0.8036879,0.010566221,0.04737707,0.017925102,0.0053889933,0.0009168827,0.00028929597,0.00036032972,0.11348823],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9960453,0.0022122636,0.00016992587,0.00052383577,0.0007873563,0.00026128092],"domain_scores_gemma":[0.9852858,0.010005931,0.00085774175,0.00086592755,0.002283839,0.0007008146],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006280491,0.00092703995,0.00041319043,0.0020574331,0.005752136,0.006614375,0.000977207,0.004072019,0.0022946661],"category_scores_gemma":[0.017608372,0.00039776292,0.000624201,0.0019707745,0.020468889,0.01227339,0.0034560796,0.0068794824,0.0006686402],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039465285,0.00009042761,0.0012551638,0.00015678584,0.000005544997,0.00015877989,0.09457277,0.00024399301,0.0006270435,0.8265991,0.040188797,0.03606212],"study_design_scores_gemma":[0.00002055409,0.0000832324,0.0030016634,0.00080950174,0.000015180662,0.0002725747,0.045643106,0.0012694857,0.0007538387,0.41625425,0.53182757,0.000049090868],"about_ca_topic_score_codex":0.002746283,"about_ca_topic_score_gemma":0.003031277,"teacher_disagreement_score":0.006614375,"about_ca_system_score_codex":0.0049386015,"about_ca_system_score_gemma":0.0023041351,"threshold_uncertainty_score":0.035832226},"labels":[],"label_agreement":null},{"id":"W3089646862","doi":"10.1016/j.cjca.2020.07.236","title":"ECG PRACTICE WITH SELF-GENERATION OF DIAGNOSES IMPROVES POST-TEST PERFORMANCE AND FLUENCY OVER MULTIPLE CHOICE","year":2020,"lang":"en","type":"article","venue":"Canadian Journal of Cardiology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Medicine; Medical diagnosis; Fluency; Interpretation (philosophy); Test (biology); Medical education; Gold standard (test); Cognitive psychology; Mathematics education; Internal medicine; Pathology; Psychology; Computer science","score_opus":0.011849256712963595,"score_gpt":0.23291653183031524,"score_spread":0.22106727511735164,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3089646862","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9949155,0.00013722814,0.0016243337,0.00017741254,0.00008224291,0.0001313118,0.00021383617,0.00034721018,0.0023709754],"genre_scores_gemma":[0.99099416,0.00017412915,0.0044067954,0.0002616607,0.000103367245,0.0001135619,0.00044678207,0.00004279367,0.003456724],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9988182,0.00043224596,0.00013790853,0.00025263047,0.00028939045,0.00006967084],"domain_scores_gemma":[0.9731758,0.020629767,0.0026996653,0.0011031012,0.0011970658,0.0011944761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002275089,0.00047471444,0.0005790369,0.00044002893,0.00018199661,0.00075988524,0.0004880041,0.0007182584,0.005225849],"category_scores_gemma":[0.024888083,0.00014877987,0.00027724504,0.00020604768,0.00016572308,0.00065788685,0.00037726422,0.0004893391,0.001881129],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.025103172,0.039868254,0.2534228,0.00060084776,0.0005711846,0.00038827033,0.0029422066,0.002911528,0.01991116,0.00021945876,0.010855597,0.64320546],"study_design_scores_gemma":[0.0021692554,0.030530324,0.9264287,0.00024842308,0.00042424473,0.00067315827,0.00078172516,0.02213908,0.010268021,0.0016227345,0.004548308,0.00016611176],"about_ca_topic_score_codex":0.0013518815,"about_ca_topic_score_gemma":0.0017772195,"teacher_disagreement_score":0.005225849,"about_ca_system_score_codex":0.00023794045,"about_ca_system_score_gemma":0.00028752408,"threshold_uncertainty_score":0.017482162},"labels":[],"label_agreement":null},{"id":"W3098677341","doi":"","title":"Categorisation techniques in computer assisted reading and analysis of texts (CARAT) in the humanities","year":2001,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Reading (process); Set (abstract data type); Computer science; Process (computing); Artificial intelligence; Natural language processing; Linguistics; Programming language; Philosophy","score_opus":0.027610245696320417,"score_gpt":0.29281819273416215,"score_spread":0.26520794703784173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3098677341","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059359516,0.0033748264,0.9775547,0.0013801442,0.00016733854,0.00052826013,0.00020445438,0.001988966,0.008865443],"genre_scores_gemma":[0.04148472,0.0015876229,0.9520608,0.00023337336,0.00016336069,0.00060767186,0.00034381432,0.0003848201,0.0031337852],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9736091,0.018326083,0.0014429153,0.002778495,0.003404368,0.00043907517],"domain_scores_gemma":[0.9611148,0.030313242,0.0014313344,0.0043084295,0.002460316,0.0003717766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016906787,0.0012801528,0.0010785629,0.014781177,0.0027111226,0.008524283,0.0029618894,0.0023752882,0.00963146],"category_scores_gemma":[0.036319442,0.0010450772,0.0016485263,0.010199745,0.008452734,0.011569714,0.005089387,0.0036350384,0.0033155303],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011511039,0.00014821727,0.0013206782,0.0014066997,0.00012614547,0.00030337527,0.015542806,0.0031921014,0.008279823,0.20207815,0.011035769,0.7564512],"study_design_scores_gemma":[0.000096955475,0.00015381933,0.0066263997,0.0013038433,0.000095494564,0.0015821436,0.0130815515,0.053442698,0.019523608,0.6569703,0.2468631,0.00026008257],"about_ca_topic_score_codex":0.0027856657,"about_ca_topic_score_gemma":0.0046478156,"teacher_disagreement_score":0.016906787,"about_ca_system_score_codex":0.0028757271,"about_ca_system_score_gemma":0.0021367618,"threshold_uncertainty_score":0.08941269},"labels":[],"label_agreement":null},{"id":"W3102769546","doi":"10.22215/etd/2020-14277","title":"A Graph-Based Indexing Technique for Efficient Searching in Large Scale Textual Documents","year":2020,"lang":"en","type":"dissertation","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Computer science; Search engine indexing; Hash table; Inverted index; Hash function; Graph; Workload; Information retrieval; Search engine; Latency (audio); Data mining; Theoretical computer science; Operating system; Programming language","score_opus":0.010708776366627721,"score_gpt":0.3233975948074325,"score_spread":0.3126888184408048,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3102769546","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03972537,0.0024856206,0.9324942,0.0006697125,0.00023780616,0.0005420628,0.002763343,0.011258135,0.009823722],"genre_scores_gemma":[0.20447148,0.002003866,0.7806152,0.00026240505,0.00015259931,0.00030758808,0.005361562,0.000674435,0.0061508585],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993673,0.00006672793,0.000054209497,0.000111250505,0.00034837838,0.000052191608],"domain_scores_gemma":[0.9983348,0.0005694006,0.00017027554,0.00042344755,0.0004358692,0.00006627702],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003860266,0.00061651133,0.0006008243,0.0034992779,0.0008468808,0.001356046,0.0011664607,0.00048246948,0.0040312563],"category_scores_gemma":[0.002708633,0.00028185477,0.00056830066,0.0068339575,0.0006260802,0.0042169876,0.00094193185,0.00076854305,0.0019966129],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040765558,0.0002627275,0.0021315366,0.001152161,0.00011130111,0.00036332474,0.0006474818,0.01946017,0.12143799,0.05979752,0.033777725,0.76045036],"study_design_scores_gemma":[0.00023098363,0.0009356725,0.0041304696,0.00019953075,0.00025250678,0.0027299176,0.0009702634,0.5562594,0.1564862,0.114512555,0.16302669,0.00026574495],"about_ca_topic_score_codex":0.0055759475,"about_ca_topic_score_gemma":0.0072652292,"teacher_disagreement_score":0.0055759475,"about_ca_system_score_codex":0.0009189371,"about_ca_system_score_gemma":0.0016097119,"threshold_uncertainty_score":0.013485968},"labels":[],"label_agreement":null},{"id":"W310333832","doi":"10.1007/978-1-4939-7131-2_352","title":"Automatic Document Topic Identification Using Social Knowledge Network","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Identification (biology); Computer science; Social knowledge; Information retrieval; Social network (sociolinguistics); Data science; World Wide Web; Social media; Sociology; Social science; Biology","score_opus":0.03248065181316784,"score_gpt":0.32177536835942844,"score_spread":0.2892947165462606,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W310333832","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036883548,0.011957904,0.9203162,0.0007295941,0.0009779383,0.00026365474,0.0025886644,0.007583688,0.018698813],"genre_scores_gemma":[0.25785092,0.0070810607,0.6799728,0.00020743781,0.0011724916,0.0003295399,0.009741441,0.0010478676,0.04259636],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992465,0.00013762993,0.000059331574,0.00021201506,0.00027680365,0.00006777701],"domain_scores_gemma":[0.99869484,0.0006353459,0.00010068577,0.00016049297,0.00035320228,0.0000554173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081746257,0.00082851324,0.00077708386,0.0050835977,0.00081317325,0.0022288307,0.0007664191,0.00080158236,0.0045073573],"category_scores_gemma":[0.0026687637,0.00034792145,0.0008280582,0.003965288,0.0003376173,0.003143952,0.0010766063,0.0009111605,0.005424949],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017141,0.00009282758,0.001467676,0.00045675779,0.00007335965,0.00016712614,0.00030553903,0.0030718746,0.03666,0.006778373,0.025587557,0.9251677],"study_design_scores_gemma":[0.00007006299,0.0002095624,0.012599321,0.00035515867,0.00046899458,0.0025028996,0.000938011,0.5907396,0.1272187,0.058902446,0.20581597,0.00017928463],"about_ca_topic_score_codex":0.001276277,"about_ca_topic_score_gemma":0.0017814747,"teacher_disagreement_score":0.0050835977,"about_ca_system_score_codex":0.0006025608,"about_ca_system_score_gemma":0.00059903605,"threshold_uncertainty_score":0.015078604},"labels":[],"label_agreement":null},{"id":"W3108390127","doi":"","title":"ScholarLensViz: A Visualization Framework for Transparency in Semantic User Profiles","year":2020,"lang":"en","type":"article","venue":"International Semantic Web Conference","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Transparency (behavior); Visualization; Data visualization; Human–computer interaction; Data mining; Computer security","score_opus":0.04154389703295188,"score_gpt":0.33338222743079854,"score_spread":0.2918383303978467,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3108390127","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0075686416,0.0002892264,0.79012895,0.00089545915,0.00023282162,0.0002951959,0.016060544,0.17493059,0.009598596],"genre_scores_gemma":[0.20856266,0.0011195287,0.70918125,0.0006808164,0.0002105312,0.0010773293,0.03219814,0.031154085,0.01581568],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986059,0.00035715758,0.00016671841,0.00019669931,0.0005292156,0.00014436149],"domain_scores_gemma":[0.99676824,0.0012061307,0.00026197854,0.0009293731,0.0005631409,0.00027119523],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025146944,0.0016380979,0.0010112913,0.005016152,0.0012929818,0.005637456,0.0016618455,0.001336495,0.019143702],"category_scores_gemma":[0.008867356,0.0008943438,0.0014319508,0.003554704,0.0006187435,0.0066372855,0.00522545,0.0026281895,0.0054973164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017766602,0.00042297164,0.006289024,0.0021473758,0.00031257264,0.00082230597,0.009650614,0.013094501,0.020801285,0.19682677,0.24191628,0.5059396],"study_design_scores_gemma":[0.0002449811,0.00012782153,0.0030532181,0.0007586024,0.0001910411,0.0004587442,0.0017275651,0.18398531,0.03923953,0.18175457,0.58815384,0.00030473305],"about_ca_topic_score_codex":0.00851051,"about_ca_topic_score_gemma":0.012996136,"teacher_disagreement_score":0.019143702,"about_ca_system_score_codex":0.001027969,"about_ca_system_score_gemma":0.0017918592,"threshold_uncertainty_score":0.06404203},"labels":[],"label_agreement":null},{"id":"W3118942295","doi":"10.3917/comla1.206.0111","title":"La loi de Zipf 70 ans après : pluridisciplinarité, modèles et controverses","year":2020,"lang":"fr","type":"article","venue":"Communication & langages","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Zipf's law; Nomination; Humanities; Philosophy; Mathematics; Political science; Law","score_opus":0.09697734801810269,"score_gpt":0.3398042514465566,"score_spread":0.24282690342845392,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3118942295","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034979574,0.2166521,0.23840824,0.37198994,0.005264773,0.0000680023,0.0014459456,0.00082184054,0.13036971],"genre_scores_gemma":[0.7092819,0.11304296,0.07366561,0.022217229,0.020037375,0.00047995456,0.00086595054,0.0011551796,0.059253804],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9946569,0.0023749096,0.00022959536,0.0012553013,0.0012439252,0.00023945488],"domain_scores_gemma":[0.9576905,0.032072876,0.0016513589,0.0032046349,0.0048101624,0.0005704585],"candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.01081474,0.0007874475,0.0015764349,0.0038785161,0.0033260144,0.010120218,0.0021579014,0.0038240855,0.007963788],"category_scores_gemma":[0.057302073,0.0008841379,0.0015161724,0.0060258745,0.012160091,0.019208036,0.0024023948,0.007708205,0.0035750074],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000023771592,0.000011772461,0.0005232305,0.00008845739,0.000019486848,0.00004300524,0.0007463536,0.0009036199,0.000054041007,0.9610811,0.015366044,0.021139193],"study_design_scores_gemma":[0.000011967109,0.000010052829,0.0009283977,0.00020849156,0.00001237555,0.00010839065,0.0003253691,0.0063323034,0.00020236064,0.91639376,0.07542843,0.00003812563],"about_ca_topic_score_codex":0.014535668,"about_ca_topic_score_gemma":0.0055496986,"teacher_disagreement_score":0.996674,"about_ca_system_score_codex":0.009840039,"about_ca_system_score_gemma":0.0033542884,"threshold_uncertainty_score":0.07139486},"labels":[],"label_agreement":null},{"id":"W3121259681","doi":"","title":"Overcoming the Invisibility of Metrology: A Reading Measurement Network for Education and the Social Sciences","year":2013,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Invisibility; Reading (process); Metrology; Traceability; Quality (philosophy); Political science; Behavioural sciences; Social science; Engineering ethics; Public relations; Sociology; Computer science; Engineering; Epistemology; Law; Mathematics; Statistics","score_opus":0.021299118856972055,"score_gpt":0.3078997637470125,"score_spread":0.28660064489004045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3121259681","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23646848,0.0036116964,0.14815484,0.4769093,0.0019276319,0.0011515927,0.000934244,0.0024156969,0.12842652],"genre_scores_gemma":[0.739269,0.0024549677,0.17899978,0.014317515,0.0010877554,0.0013980157,0.001596648,0.000636287,0.060240105],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98303324,0.01163842,0.0006743783,0.0015736076,0.0022805159,0.0007997396],"domain_scores_gemma":[0.8819088,0.067898475,0.005933737,0.01150183,0.018179717,0.014577482],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.041906197,0.0003093433,0.00039522443,0.0038005963,0.007391155,0.007504477,0.0024431078,0.00471599,0.012259508],"category_scores_gemma":[0.08286928,0.0003930733,0.0003392256,0.0034680201,0.0035187572,0.017075144,0.011490902,0.0038228869,0.0026796178],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031205363,0.0006134217,0.04811377,0.00033282273,0.000041535546,0.0010563014,0.036786284,0.0013823962,0.0031877283,0.17375514,0.095317625,0.63910085],"study_design_scores_gemma":[0.000089343775,0.00047632292,0.03006255,0.0007174652,0.00005021678,0.00065579254,0.029921265,0.015683096,0.0021363872,0.1881178,0.7319507,0.00013908418],"about_ca_topic_score_codex":0.0039984146,"about_ca_topic_score_gemma":0.0056591057,"teacher_disagreement_score":0.041906197,"about_ca_system_score_codex":0.0040262165,"about_ca_system_score_gemma":0.014668809,"threshold_uncertainty_score":0.22162378},"labels":[],"label_agreement":null},{"id":"W3122916539","doi":"","title":"Mind the Gap: Accounting for Measurement Error and Misclassification in Variables Generated via Data Mining","year":2017,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Quest University Canada","funders":"","keywords":"Econometrics; Computer science; Econometric model; Covariate; Data mining; Errors-in-variables models; Observational error; Variance (accounting); Inference; Parameterized complexity; Statistics; Machine learning; Artificial intelligence; Algorithm; Mathematics; Accounting; Economics","score_opus":0.09644917981351636,"score_gpt":0.33130341012797304,"score_spread":0.23485423031445668,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3122916539","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049100243,0.006015672,0.92577314,0.010656801,0.001346191,0.00047071785,0.0007623681,0.0005544704,0.0053204363],"genre_scores_gemma":[0.67582506,0.0024927023,0.3110394,0.0051721516,0.0013126833,0.0012104678,0.0010245539,0.00027459252,0.0016484103],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.8789345,0.08683025,0.008519769,0.012094889,0.0122571,0.0013634815],"domain_scores_gemma":[0.4402853,0.46496874,0.030559242,0.042446796,0.020761564,0.0009783471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.13021801,0.001703449,0.0019098006,0.005841204,0.0022029905,0.0051571783,0.0048883297,0.0039914646,0.0014054914],"category_scores_gemma":[0.5256914,0.0009274043,0.0024489893,0.007529268,0.004928608,0.010241195,0.0055140443,0.0051362864,0.00044299877],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045832578,0.00020704258,0.23742707,0.002097576,0.0028604511,0.0007399239,0.0090701245,0.07630512,0.0006446602,0.2909547,0.016026108,0.36320895],"study_design_scores_gemma":[0.00012368613,0.00029324525,0.047093872,0.0034899455,0.00086589233,0.0008057123,0.0026436208,0.33601356,0.0032384004,0.57165587,0.033528104,0.0002481442],"about_ca_topic_score_codex":0.008540502,"about_ca_topic_score_gemma":0.006147786,"teacher_disagreement_score":0.13021801,"about_ca_system_score_codex":0.0027967987,"about_ca_system_score_gemma":0.0045223697,"threshold_uncertainty_score":0.68866694},"labels":[],"label_agreement":null},{"id":"W3123195249","doi":"","title":"Contrasting Rule-Based and Similarity-Based Category Learning: The Effects of Mood and Prior Knowledge on Ambiguous Categorization","year":2011,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; York University","funders":"","keywords":"Categorization; Similarity (geometry); Product (mathematics); Feature (linguistics); Product category; Concept learning; Mood; Phenomenon; Cognitive psychology; Psychology; Artificial intelligence; Computer science; Natural language processing; Social psychology; Mathematics; Linguistics; Epistemology","score_opus":0.013585126114174633,"score_gpt":0.24112069504931413,"score_spread":0.2275355689351395,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3123195249","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99139345,0.000115132454,0.0027221215,0.00015126921,0.000016607706,0.00004053391,0.00003465056,0.000014802055,0.0055114506],"genre_scores_gemma":[0.9975038,0.000060450493,0.001756422,0.00011884822,0.000024157644,0.000025840318,0.000050764702,0.000017493972,0.00044219612],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9968425,0.0015624586,0.00017169573,0.00064955995,0.0006203064,0.00015349341],"domain_scores_gemma":[0.9010328,0.08128989,0.0074568694,0.0059946203,0.002701866,0.0015241075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049898373,0.0003066518,0.0005839106,0.0005411423,0.0005192717,0.002625054,0.00062082487,0.0008687319,0.002991809],"category_scores_gemma":[0.066553935,0.00044741307,0.00045870975,0.00029982807,0.0012905439,0.0018661168,0.0013669527,0.0016615553,0.00043705915],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.025472112,0.007176673,0.38287786,0.0008318177,0.001062829,0.0007409313,0.013970677,0.0068430705,0.29925206,0.0094755255,0.0013081089,0.25098836],"study_design_scores_gemma":[0.00067379687,0.0035925931,0.91219044,0.00009376157,0.0006845682,0.00059248286,0.0020730835,0.027565643,0.0295341,0.021523217,0.001251791,0.00022455817],"about_ca_topic_score_codex":0.0012370875,"about_ca_topic_score_gemma":0.0017384058,"teacher_disagreement_score":0.0049898373,"about_ca_system_score_codex":0.0005732515,"about_ca_system_score_gemma":0.00030431026,"threshold_uncertainty_score":0.026389062},"labels":[],"label_agreement":null},{"id":"W3124677217","doi":"","title":"Information about information: a taxonomy of views","year":2010,"lang":"en","type":"article","venue":"MIS Quarterly","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Knowledge management; Taxonomy (biology); Information system; Computer science; Data science; Information retrieval; Business; Political science","score_opus":0.010706269992895189,"score_gpt":0.24259368547888452,"score_spread":0.23188741548598932,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3124677217","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09486475,0.057464276,0.46911782,0.02654799,0.0008773092,0.00068224507,0.0046714046,0.0012121098,0.34456214],"genre_scores_gemma":[0.75033593,0.04975844,0.1780986,0.0021141437,0.001212269,0.0006571197,0.0042342055,0.00039555968,0.013193738],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9919457,0.002667093,0.00081789005,0.0005845492,0.003472984,0.00051176344],"domain_scores_gemma":[0.9717159,0.018041192,0.002010228,0.002886465,0.0043072524,0.0010388718],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052332804,0.00071994384,0.0007375361,0.017523706,0.002075106,0.015990198,0.0015020008,0.0025480315,0.005294778],"category_scores_gemma":[0.031220762,0.0007936523,0.0014331813,0.01747734,0.005426331,0.044170223,0.0040886053,0.0027036117,0.0013353124],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008265459,0.000039458682,0.0063946,0.00068111136,0.00006190023,0.0003235362,0.012494058,0.0006078175,0.0012634133,0.8647997,0.007347207,0.10590457],"study_design_scores_gemma":[0.000031192925,0.00010606711,0.0037388876,0.0018666335,0.00019328004,0.0017229319,0.017243207,0.008437614,0.001767347,0.7774363,0.18737285,0.00008366013],"about_ca_topic_score_codex":0.0020423925,"about_ca_topic_score_gemma":0.0013007071,"teacher_disagreement_score":0.017523706,"about_ca_system_score_codex":0.0021370945,"about_ca_system_score_gemma":0.0022883979,"threshold_uncertainty_score":0.027676523},"labels":[],"label_agreement":null},{"id":"W3125334212","doi":"10.17705/1jais.00237","title":"A Theory-Driven Design Framework for Social Recommender Systems","year":2010,"lang":"en","type":"article","venue":"Journal of the Association for Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":138,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; City University of New York","keywords":"Recommender system; Computer science; Competence (human resources); Similarity (geometry); Designtheory; Design science; Artificial intelligence; Information retrieval; Knowledge management; Human–computer interaction; Psychology; Social psychology","score_opus":0.026222151389128694,"score_gpt":0.3018622921912353,"score_spread":0.2756401408021066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3125334212","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011470118,0.00012927716,0.99471587,0.00070137024,0.00004173315,0.00059518334,0.00004274834,0.000117226715,0.0025095092],"genre_scores_gemma":[0.04326951,0.0002338052,0.9519433,0.0002316775,0.00004225729,0.0029255839,0.00011355474,0.000036599664,0.0012036896],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98461443,0.009619195,0.001032521,0.0012917977,0.0030373135,0.0004047191],"domain_scores_gemma":[0.98555654,0.008594142,0.0008917573,0.0015834968,0.0027908469,0.0005833025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018976651,0.0023196274,0.0012882414,0.00288882,0.0017308532,0.004889253,0.004368645,0.0038421347,0.0052688583],"category_scores_gemma":[0.019696549,0.0015204027,0.0026369973,0.0016602151,0.0045311023,0.0032730354,0.0035383587,0.0035709331,0.0016851288],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001034649,0.00034284408,0.0012743439,0.0012502944,0.00024025839,0.0004943987,0.0024409513,0.12758206,0.004246604,0.7918071,0.0027722584,0.06744541],"study_design_scores_gemma":[0.00031405568,0.00051278865,0.00034340957,0.00048940646,0.00018422738,0.0003470029,0.00062854425,0.46159163,0.0028100577,0.4819431,0.050727382,0.000108316046],"about_ca_topic_score_codex":0.0035151308,"about_ca_topic_score_gemma":0.0049862023,"teacher_disagreement_score":0.018976651,"about_ca_system_score_codex":0.0033937842,"about_ca_system_score_gemma":0.0060458314,"threshold_uncertainty_score":0.10035926},"labels":[],"label_agreement":null},{"id":"W3129531092","doi":"10.1109/icdmw51313.2020.00088","title":"Graph-based Topic Extraction Using Centroid Distance of Phrase Embeddings on Healthy Aging Open-ended Survey Questions","year":2020,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Natural language processing; Phrase; Artificial intelligence; Information retrieval; Word (group theory); Graph; Centroid; Task (project management); Domain (mathematical analysis); Linguistics","score_opus":0.06804492174385719,"score_gpt":0.3786420053132639,"score_spread":0.31059708356940674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3129531092","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16328411,0.002458385,0.8185581,0.0005298259,0.0003173078,0.0004917372,0.0057483055,0.0054190955,0.0031930495],"genre_scores_gemma":[0.61740077,0.001116025,0.35295597,0.00014855826,0.0003323849,0.00066820753,0.021556592,0.00042976593,0.005391714],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986545,0.00046775196,0.00011809958,0.00043674454,0.000188994,0.0001338172],"domain_scores_gemma":[0.9966589,0.0020355273,0.00029574046,0.00027749914,0.0006275147,0.00010478426],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012889465,0.001194428,0.00086320343,0.005025094,0.00040575804,0.001274821,0.00077927136,0.0010619462,0.0024394905],"category_scores_gemma":[0.0063730595,0.00025098212,0.0010822127,0.0038960641,0.00034162856,0.0018296335,0.0010886468,0.0008125807,0.002789462],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010486619,0.00029770628,0.010073123,0.00091245055,0.00022517313,0.00034440763,0.0014238487,0.028963333,0.021008637,0.006829649,0.021867625,0.90700537],"study_design_scores_gemma":[0.00015006996,0.00036915098,0.017467389,0.00012557795,0.00016010157,0.00046757623,0.0017152951,0.9257773,0.011683389,0.024198888,0.017789153,0.00009602005],"about_ca_topic_score_codex":0.003268841,"about_ca_topic_score_gemma":0.0043073418,"teacher_disagreement_score":0.005025094,"about_ca_system_score_codex":0.00057036546,"about_ca_system_score_gemma":0.0007095957,"threshold_uncertainty_score":0.008160949},"labels":[],"label_agreement":null},{"id":"W3149495681","doi":"10.18280/isi.260112","title":"Extractive Text Summarization Using Recent Approaches: A Survey","year":2021,"lang":"en","type":"article","venue":"Ingénierie des systèmes d information","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Information retrieval; Text graph; Multi-document summarization; Process (computing); Representation (politics); Natural language processing; Artificial intelligence; Data science","score_opus":0.06615902636918873,"score_gpt":0.27934401843601653,"score_spread":0.21318499206682778,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3149495681","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010353289,0.440272,0.5295012,0.0017874388,0.0012649673,0.0007057584,0.0012844808,0.0043490445,0.010481852],"genre_scores_gemma":[0.0693597,0.46161234,0.4450122,0.0012117043,0.0047043497,0.0009052011,0.006329567,0.0010656776,0.009799169],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99615943,0.00078741123,0.0006228971,0.0009544255,0.0013636968,0.000112052796],"domain_scores_gemma":[0.99136883,0.004651628,0.00084381655,0.0007569168,0.0022313083,0.00014744059],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026602547,0.0027221967,0.0022425747,0.009750643,0.00090510753,0.0029106047,0.002344042,0.0012871737,0.0036686205],"category_scores_gemma":[0.009844849,0.0008512106,0.0021909643,0.01245825,0.00093768974,0.0052676937,0.0012280593,0.0014830078,0.0033541338],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007761697,0.00009625837,0.00047628794,0.005250427,0.0001547908,0.00008974963,0.00037697607,0.0023242082,0.0048071514,0.003129962,0.009928327,0.9732881],"study_design_scores_gemma":[0.000164173,0.0015465333,0.010420276,0.0042466256,0.00196185,0.0034851397,0.0029125405,0.07991147,0.05532223,0.035342798,0.80424505,0.00044132638],"about_ca_topic_score_codex":0.002008402,"about_ca_topic_score_gemma":0.0015836873,"teacher_disagreement_score":0.009750643,"about_ca_system_score_codex":0.0007197994,"about_ca_system_score_gemma":0.0013646588,"threshold_uncertainty_score":0.014068961},"labels":[],"label_agreement":null},{"id":"W3152040157","doi":"10.22148/001c.22086","title":"An Institutional Perspective on Genres: Generic Subtitles in German Literature from 1500-2020","year":2021,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Perspective (graphical); German; Relation (database); Computer science; Poetry; Field (mathematics); Linguistics; Artificial intelligence; Philosophy","score_opus":0.018246796622688802,"score_gpt":0.32749123006626346,"score_spread":0.30924443344357466,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3152040157","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9393314,0.0060044457,0.0024895382,0.00047987397,0.000038704646,0.00007717448,0.032529887,0.0001242858,0.01892452],"genre_scores_gemma":[0.9708803,0.0023778104,0.0026720564,0.000050944927,0.000027996648,0.000081262544,0.02069962,0.000042941843,0.0031670507],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9990845,0.00011148556,0.000154126,0.00024562393,0.00026071607,0.00014354552],"domain_scores_gemma":[0.9950334,0.0014761738,0.0018023249,0.00048582852,0.0009211171,0.00028127417],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013623727,0.00026204114,0.00029029918,0.022236777,0.00056482654,0.0033634843,0.0004991443,0.00033827938,0.0031194345],"category_scores_gemma":[0.005649991,0.00018224928,0.00024757563,0.026470494,0.00095074804,0.002562289,0.0016261456,0.00033971673,0.0011685954],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042614265,0.00010902143,0.5374836,0.0021232842,0.00021164306,0.0018737245,0.048922047,0.0033285732,0.0058747595,0.034714278,0.01902381,0.34590915],"study_design_scores_gemma":[0.000007464709,0.000050172093,0.8674776,0.00040611424,0.000078986144,0.0005511357,0.017497042,0.001366903,0.0019836288,0.0037176197,0.10682167,0.000041721258],"about_ca_topic_score_codex":0.014783873,"about_ca_topic_score_gemma":0.028331181,"teacher_disagreement_score":0.022236777,"about_ca_system_score_codex":0.0019133986,"about_ca_system_score_gemma":0.0008478537,"threshold_uncertainty_score":0.02939564},"labels":[],"label_agreement":null},{"id":"W3160310472","doi":"10.32473/flairs.v34i1.128502","title":"Multilingual Automatic Term Extraction in Low-Resource Domains","year":2021,"lang":"en","type":"article","venue":"Proceedings of the ... International Florida Artificial Intelligence Research Society Conference","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Task (project management); Term (time); Artificial intelligence; Resource (disambiguation); Raw data; Sequence labeling; Sequence (biology); Domain (mathematical analysis); Artificial neural network; Natural language processing; Deep learning; Information extraction; Machine learning; Engineering","score_opus":0.08710697055732607,"score_gpt":0.4004800386628963,"score_spread":0.3133730681055702,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3160310472","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20226511,0.00913743,0.73277193,0.0016006863,0.0007493964,0.0004283052,0.01931335,0.018474037,0.015259748],"genre_scores_gemma":[0.42197356,0.0021296535,0.50417966,0.00043762728,0.00038323362,0.00030578944,0.057861447,0.000926097,0.01180298],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99862194,0.00041567357,0.0001635563,0.00038789035,0.0002534913,0.0001574765],"domain_scores_gemma":[0.99728763,0.0012332903,0.00020949575,0.00055323465,0.0006095677,0.00010683814],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014867703,0.0012703617,0.001111637,0.0041978345,0.0008291696,0.0015851264,0.0009441678,0.00094927364,0.003077663],"category_scores_gemma":[0.0049543004,0.00036007137,0.00091934827,0.0036824096,0.0005087783,0.004881974,0.0026482642,0.001719843,0.004027167],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00087747566,0.00042202082,0.0046329875,0.0013447693,0.00020168682,0.0010490182,0.0005134538,0.019794473,0.087259874,0.008487034,0.04245533,0.83296186],"study_design_scores_gemma":[0.0002441867,0.00045263983,0.014271164,0.00030286828,0.00032239975,0.0019099322,0.0016829643,0.63913107,0.14521137,0.065517046,0.13070323,0.00025120436],"about_ca_topic_score_codex":0.002887609,"about_ca_topic_score_gemma":0.00632849,"teacher_disagreement_score":0.0041978345,"about_ca_system_score_codex":0.0005839245,"about_ca_system_score_gemma":0.0012654599,"threshold_uncertainty_score":0.010295808},"labels":[],"label_agreement":null},{"id":"W3165419854","doi":"10.5281/zenodo.4622059","title":"Detecting Character References in Literary Novels using a Two Stage Contextual Deep Learning approach","year":2019,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Character (mathematics); Artificial intelligence; Linguistics; Natural language processing; Deep learning; Computer science; Psychology; Literature; History; Art; Philosophy; Mathematics","score_opus":0.04695709435782621,"score_gpt":0.2825374789226426,"score_spread":0.23558038456481636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3165419854","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7221998,0.011484865,0.20525365,0.0018114485,0.0018851612,0.00044168512,0.013393949,0.01000107,0.0335283],"genre_scores_gemma":[0.87597793,0.0013878498,0.08706633,0.00020790516,0.0007063149,0.00018879374,0.014024108,0.000387404,0.020053416],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991775,0.00013794091,0.00005636352,0.00033927118,0.00018037856,0.00010856978],"domain_scores_gemma":[0.99810994,0.00072866335,0.0002258871,0.0001987524,0.0005653544,0.00017136062],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047828906,0.0007140314,0.0005114046,0.0044290298,0.00095270626,0.0016597614,0.000907545,0.0011282508,0.0054439465],"category_scores_gemma":[0.0033364345,0.0002620891,0.000514043,0.0025800145,0.00043083934,0.0018655059,0.0016085468,0.0010011737,0.0050203805],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015673268,0.0004983556,0.02355354,0.001405456,0.00021188658,0.00255118,0.0019887625,0.006302278,0.06970246,0.0044213394,0.060538184,0.8272591],"study_design_scores_gemma":[0.00015518522,0.001025495,0.10997336,0.0006558662,0.00042264853,0.0046386095,0.008486355,0.59067,0.09070422,0.021313228,0.1717449,0.00021023948],"about_ca_topic_score_codex":0.0019079627,"about_ca_topic_score_gemma":0.0055609182,"teacher_disagreement_score":0.0054439465,"about_ca_system_score_codex":0.00046720845,"about_ca_system_score_gemma":0.00038962634,"threshold_uncertainty_score":0.018211842},"labels":[],"label_agreement":null},{"id":"W3168080743","doi":"10.1142/s0129183121501448","title":"Are all the word ranking methods the same?","year":2021,"lang":"en","type":"article","venue":"International Journal of Modern Physics C","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Iran National Science Foundation","keywords":"Ranking (information retrieval); Word (group theory); Rank (graph theory); Computer science; Rank correlation; Artificial intelligence; Natural language processing; Sample (material); Information retrieval; Mathematics; Machine learning; Combinatorics; Physics","score_opus":0.051366036786578244,"score_gpt":0.38911578411534226,"score_spread":0.337749747328764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3168080743","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10957373,0.050172925,0.76771426,0.023857571,0.0044343984,0.0009645676,0.0030147107,0.0036079178,0.036659848],"genre_scores_gemma":[0.46734017,0.015409816,0.49303496,0.004901689,0.0040688436,0.00087572855,0.0025003683,0.0016286662,0.010239842],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9760157,0.011715281,0.0012303905,0.0040353765,0.0062359697,0.0007672045],"domain_scores_gemma":[0.9090504,0.047789246,0.007745364,0.012882738,0.020899316,0.0016329583],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02704585,0.0019880447,0.0025368289,0.0067439354,0.0015990485,0.00906509,0.0027073193,0.002729818,0.0059610684],"category_scores_gemma":[0.1266339,0.00084754475,0.0018140294,0.0070447773,0.0026294373,0.015231789,0.0015480614,0.0031149003,0.008700002],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007590769,0.0002618824,0.030464606,0.0018485369,0.0020606506,0.00010607194,0.0014127475,0.0058838953,0.0034669258,0.037336353,0.023497218,0.892902],"study_design_scores_gemma":[0.00048489627,0.0022176746,0.108470336,0.003223837,0.003010412,0.0023875663,0.006551228,0.24254182,0.017848952,0.48839802,0.12341863,0.0014466833],"about_ca_topic_score_codex":0.0033159447,"about_ca_topic_score_gemma":0.0030015009,"teacher_disagreement_score":0.02704585,"about_ca_system_score_codex":0.0011642172,"about_ca_system_score_gemma":0.002158612,"threshold_uncertainty_score":0.1430338},"labels":[],"label_agreement":null},{"id":"W3176967269","doi":"","title":"TREC 2020 Podcasts Track Overview.","year":2020,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Track (disk drive); Computer science; Information retrieval; Operating system","score_opus":0.06050055887350622,"score_gpt":0.310150801677166,"score_spread":0.2496502428036598,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176967269","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004341382,0.024673682,0.024273464,0.014994732,0.0485831,0.00419523,0.63831353,0.048343476,0.19228138],"genre_scores_gemma":[0.006311336,0.007253632,0.018531969,0.0032397024,0.008082471,0.0021956535,0.6154247,0.0050389213,0.33392158],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9951609,0.00079144456,0.00024062836,0.00037775966,0.0027003703,0.0007290249],"domain_scores_gemma":[0.9749223,0.0020292988,0.0012171979,0.0016046065,0.0146441525,0.0055823955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0118053965,0.0042334264,0.002574337,0.01184147,0.0032286076,0.009065449,0.0059378454,0.0029471486,0.16026507],"category_scores_gemma":[0.014582235,0.0014381435,0.0017970599,0.009026304,0.00094726885,0.0069361217,0.004330505,0.0035045694,0.20752099],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000067346315,0.000028084592,0.000054173408,0.00023309237,0.000009677364,0.000009274662,0.000007197609,0.000100453144,0.0005116216,0.0001541133,0.9872155,0.01160951],"study_design_scores_gemma":[0.00021232638,0.00020004435,0.002180599,0.00031013362,0.00007259021,0.000080075864,0.00007686051,0.001422332,0.0029015045,0.0013469792,0.9911097,0.00008689659],"about_ca_topic_score_codex":0.11912479,"about_ca_topic_score_gemma":0.18345468,"teacher_disagreement_score":0.16026507,"about_ca_system_score_codex":0.0045192153,"about_ca_system_score_gemma":0.017015368,"threshold_uncertainty_score":0.5361401},"labels":[],"label_agreement":null},{"id":"W3177106525","doi":"","title":"The DELICES project: Indexing scientific literature through semantic expansion","year":2020,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Search engine indexing; Information retrieval; Natural language processing","score_opus":0.018038087111500255,"score_gpt":0.2554298647587292,"score_spread":0.23739177764722894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3177106525","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06260635,0.029510751,0.47397432,0.0068812305,0.0033410026,0.00394618,0.29825956,0.059831664,0.061648957],"genre_scores_gemma":[0.064573064,0.00893597,0.5663857,0.00090022647,0.0009751943,0.0016738543,0.34162298,0.0037215797,0.011211334],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9948037,0.0013078497,0.0007911092,0.0010129598,0.0018707017,0.00021360243],"domain_scores_gemma":[0.9919344,0.0037305052,0.0007315784,0.0012506458,0.0017249862,0.0006277803],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0037816092,0.002375713,0.002094204,0.029190127,0.0017143764,0.004419681,0.0019303884,0.0013551422,0.01665077],"category_scores_gemma":[0.015811823,0.00079269544,0.0020748125,0.017731715,0.0016316273,0.0071859695,0.0063918144,0.0016077467,0.009185141],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011512334,0.0004730726,0.003076053,0.009810807,0.00064652227,0.00065669476,0.0015380472,0.0042687934,0.033911526,0.05312952,0.26219332,0.6291445],"study_design_scores_gemma":[0.0007314759,0.0005204948,0.0069673453,0.0019471361,0.0010957749,0.0011032629,0.0021436745,0.04425206,0.026524168,0.093859926,0.82058537,0.00026931523],"about_ca_topic_score_codex":0.004412038,"about_ca_topic_score_gemma":0.005604602,"teacher_disagreement_score":0.9955803,"about_ca_system_score_codex":0.0013120945,"about_ca_system_score_gemma":0.0043459316,"threshold_uncertainty_score":0.05570239},"labels":[],"label_agreement":null},{"id":"W3188655801","doi":"10.1037/cbs0000294","title":"TMI? Accompanying details impact statements’ perceived veracity.","year":2021,"lang":"en","type":"article","venue":"Canadian Journal of Behavioural Science/Revue canadienne des sciences du comportement","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Psychology; Applied psychology; Social psychology","score_opus":0.10438393626355984,"score_gpt":0.328269907150494,"score_spread":0.22388597088693413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3188655801","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011419435,0.0005499122,0.004074549,0.0060397913,0.0033203347,0.001446691,0.70476025,0.008623323,0.25976565],"genre_scores_gemma":[0.14915739,0.0020551013,0.031944275,0.0045645274,0.0023591623,0.00823174,0.4835976,0.0088389935,0.3092511],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99817383,0.00040184718,0.00024012377,0.00010789475,0.0008969194,0.00017936446],"domain_scores_gemma":[0.9504221,0.032238245,0.003055527,0.0020193586,0.010862568,0.0014022173],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0031990814,0.0005983519,0.0005822728,0.0034874189,0.00076689577,0.001786147,0.0010106268,0.00089833385,0.71190906],"category_scores_gemma":[0.058249425,0.00026094986,0.00039755233,0.0050257784,0.0003223025,0.0027164628,0.00179572,0.0012596427,0.19286273],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013883015,0.00006710398,0.0015070942,0.0007134893,0.0000065213903,0.000023023436,0.00024644812,0.000061996034,0.00016642358,0.00079763116,0.9585666,0.037704892],"study_design_scores_gemma":[0.0001411143,0.00010811924,0.038716648,0.0012635064,0.000034907192,0.00017629315,0.0025464424,0.00070305704,0.0012810448,0.006145111,0.9487811,0.00010263229],"about_ca_topic_score_codex":0.0048207184,"about_ca_topic_score_gemma":0.0077314693,"teacher_disagreement_score":0.71190906,"about_ca_system_score_codex":0.0008018688,"about_ca_system_score_gemma":0.0010438164,"threshold_uncertainty_score":0.41092676},"labels":[],"label_agreement":null},{"id":"W3196840400","doi":"10.2139/ssrn.3908201","title":"Comparing Methods of Exploratory Data Analysis for the Moral Foundations Questionnaire with a Small Sample","year":2021,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Northern Alberta Institute of Technology","funders":"","keywords":"Sample (material); Psychology; Chromatography","score_opus":0.11276888888840333,"score_gpt":0.3753692400134184,"score_spread":0.2626003511250151,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196840400","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55027944,0.0014405032,0.40644315,0.0008515853,0.00064269605,0.032938875,0.0010460921,0.00063104846,0.0057266946],"genre_scores_gemma":[0.58323556,0.00040360278,0.32183325,0.0005407257,0.00015457024,0.091128714,0.00076296733,0.0005953809,0.0013451682],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.5973598,0.36082238,0.016708842,0.010135889,0.012949913,0.0020232387],"domain_scores_gemma":[0.13483772,0.82238734,0.008049106,0.018590204,0.0153388465,0.0007968019],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2419734,0.0019630417,0.0030487652,0.0042510247,0.0034754206,0.003913195,0.0033038447,0.0029580828,0.008062754],"category_scores_gemma":[0.6066621,0.0017672805,0.005424522,0.0034367756,0.003322382,0.0061659436,0.0046088872,0.0033616198,0.0011237715],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.06231323,0.015104851,0.1719138,0.013188742,0.012899404,0.0013029812,0.16563128,0.010379011,0.012150543,0.032016538,0.0124552855,0.4906444],"study_design_scores_gemma":[0.04629416,0.08593439,0.40628618,0.0066112797,0.014257178,0.002612555,0.13459432,0.13539043,0.031051941,0.099824406,0.034871675,0.002271451],"about_ca_topic_score_codex":0.0033483221,"about_ca_topic_score_gemma":0.0058926526,"teacher_disagreement_score":0.2419734,"about_ca_system_score_codex":0.003246374,"about_ca_system_score_gemma":0.0048451405,"threshold_uncertainty_score":0.93478173},"labels":[],"label_agreement":null},{"id":"W3200955930","doi":"10.1109/access.2021.3111833","title":"Senti-COVID19: An Interactive Visual Analytics System for Detecting Public Sentiment and Insights Regarding COVID-19 From Social Media","year":2021,"lang":"en","type":"article","venue":"IEEE Access","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Sentiment analysis; Social media; Computer science; Coronavirus disease 2019 (COVID-19); Lexicon; Public opinion; Action (physics); Data science; Social media analytics; Artificial intelligence; World Wide Web; Political science; Medicine","score_opus":0.07433494689166144,"score_gpt":0.38851405342389667,"score_spread":0.31417910653223524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3200955930","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06262157,0.00096911856,0.42452914,0.0016551521,0.0008822503,0.00196701,0.10012528,0.3648104,0.042440064],"genre_scores_gemma":[0.36042663,0.0013061415,0.49665862,0.0013166124,0.0004580159,0.0035186873,0.088708825,0.01994908,0.027657352],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99953413,0.000103847895,0.000045149773,0.000102247825,0.00016743928,0.000047199243],"domain_scores_gemma":[0.9984308,0.00079621264,0.00015093359,0.00014462194,0.000355708,0.00012172768],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011241785,0.0014533963,0.00065600325,0.0034017575,0.00066641386,0.0018918064,0.0010846365,0.0007054563,0.021625642],"category_scores_gemma":[0.0048526805,0.00041437065,0.00088905945,0.0014928593,0.0003604031,0.002791763,0.0025948628,0.0008258007,0.0049274154],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002428261,0.000343488,0.01010993,0.0030322773,0.00038321866,0.0014145793,0.0068720076,0.005149864,0.06888841,0.016206272,0.52788323,0.35728857],"study_design_scores_gemma":[0.0004569487,0.0006049049,0.02420175,0.0006183073,0.00029853784,0.0011058124,0.0029986643,0.3193871,0.057586707,0.054333203,0.53785455,0.00055342924],"about_ca_topic_score_codex":0.0036124054,"about_ca_topic_score_gemma":0.0051770527,"teacher_disagreement_score":0.021625642,"about_ca_system_score_codex":0.0006617766,"about_ca_system_score_gemma":0.00068271364,"threshold_uncertainty_score":0.07234496},"labels":[],"label_agreement":null},{"id":"W3201804497","doi":"","title":"A Basic Morphological Parser for Discourse Information Grammar","year":2004,"lang":"en","type":"article","venue":"Papers from the Annual Meetings of the Atlantic Provinces Linguistic Association (PAMAPLA)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Perspective (graphical); Parsing; Lexicon; Grammar; Meaning (existential); Artificial intelligence; Natural language processing; Linguistics; Rule-based machine translation; Epistemology","score_opus":0.006281533642535385,"score_gpt":0.23727348965213016,"score_spread":0.23099195600959477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3201804497","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018503997,0.00021288455,0.95209247,0.0010085118,0.00024510926,0.00025941982,0.0039454377,0.024305679,0.01608011],"genre_scores_gemma":[0.05630644,0.0005191267,0.91040933,0.0006902319,0.00026465664,0.00040664678,0.006022119,0.009113174,0.016268315],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987919,0.00028229083,0.00016126284,0.00038003374,0.0002863703,0.00009824145],"domain_scores_gemma":[0.9982317,0.00067458814,0.000103606704,0.000491787,0.0004358138,0.00006249574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020336758,0.0009613082,0.0009629388,0.002288978,0.0018780142,0.004316362,0.0019511192,0.0020622734,0.03714582],"category_scores_gemma":[0.0060239015,0.0015161638,0.0016489658,0.0022245285,0.0018701268,0.00703532,0.0028748016,0.0028970896,0.020119065],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017235123,0.00010142891,0.001479623,0.0005692585,0.000071461815,0.0008118931,0.0018212644,0.0032258523,0.01701275,0.5984667,0.07230113,0.30396622],"study_design_scores_gemma":[0.00004669199,0.000063250634,0.0010046854,0.00024012329,0.0000672555,0.0013177493,0.00042514916,0.04138066,0.01729775,0.5663796,0.37167007,0.00010690868],"about_ca_topic_score_codex":0.0021308914,"about_ca_topic_score_gemma":0.0026839771,"teacher_disagreement_score":0.03714582,"about_ca_system_score_codex":0.0012375383,"about_ca_system_score_gemma":0.0025573545,"threshold_uncertainty_score":0.124265134},"labels":[],"label_agreement":null},{"id":"W3210301698","doi":"","title":"A Semantic Metadata Enrichment Software Ecosystem Based on Topic Metadata Enrichment","year":2017,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Metadata; Computer science; Information retrieval; Semantic grid; Scalability; Metadata modeling; World Wide Web; Annotation; Linked data; Metadata repository; Semantic Web; Database; Artificial intelligence","score_opus":0.01417168031758521,"score_gpt":0.27847943595544383,"score_spread":0.26430775563785863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3210301698","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030527174,0.00024836275,0.9259158,0.00037552227,0.000071695176,0.00045039557,0.0006626421,0.037345767,0.004402692],"genre_scores_gemma":[0.08115312,0.00020389644,0.9105363,0.00016150285,0.00003505351,0.00022397575,0.0025344137,0.0009501759,0.0042015323],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99867105,0.00022459852,0.00020641636,0.00028022093,0.0005539226,0.00006374741],"domain_scores_gemma":[0.9967514,0.0010479015,0.00020081847,0.0008548164,0.00095562317,0.00018955246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025200108,0.0006001413,0.00075553195,0.005000016,0.0011138406,0.002325073,0.0012625001,0.0009232549,0.002393576],"category_scores_gemma":[0.005867864,0.00064307154,0.0013386639,0.0031674195,0.0007326217,0.0050713248,0.0032571817,0.0011517627,0.001867509],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075578335,0.0007310201,0.020646103,0.00089562865,0.00036201757,0.00091099687,0.002076045,0.011735993,0.06692828,0.038533207,0.016935334,0.8394896],"study_design_scores_gemma":[0.00026975886,0.00055899174,0.010851416,0.00029577073,0.00042951567,0.0029457076,0.0010224036,0.583384,0.14841117,0.05323666,0.19822836,0.00036613134],"about_ca_topic_score_codex":0.0024412153,"about_ca_topic_score_gemma":0.0038556657,"teacher_disagreement_score":0.005000016,"about_ca_system_score_codex":0.00071064907,"about_ca_system_score_gemma":0.0015636799,"threshold_uncertainty_score":0.013327241},"labels":[],"label_agreement":null},{"id":"W3214508694","doi":"10.1002/mus.27456","title":"Combining multiple measures into a summary index: A step toward more reliable measurement","year":2021,"lang":"en","type":"letter","venue":"Muscle & Nerve","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Index (typography); Computer science; Statistics; Psychology; Mathematics; World Wide Web","score_opus":0.046154923928692346,"score_gpt":0.26447309893444293,"score_spread":0.21831817500575057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3214508694","genre_codex":"commentary","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017828704,0.022047544,0.04876824,0.86293584,0.053339235,0.00028086404,0.00089070405,0.0006785935,0.009275995],"genre_scores_gemma":[0.050147977,0.029212095,0.2237904,0.4815912,0.19232494,0.0018664805,0.0017530216,0.001540708,0.017773261],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.94257784,0.028155057,0.007895411,0.0024537004,0.01824787,0.00067011],"domain_scores_gemma":[0.75995994,0.14474481,0.01288522,0.012579077,0.06669154,0.0031393906],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07145466,0.0012455164,0.0030362096,0.0047744517,0.0017233128,0.008101911,0.0034227853,0.0076802047,0.006787643],"category_scores_gemma":[0.20937356,0.00066609296,0.0014634792,0.004009627,0.0030357188,0.009547167,0.003162037,0.01649311,0.009265045],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014241031,0.0001003506,0.006104336,0.00083434995,0.0002067092,0.00016415684,0.000430955,0.00012420525,0.00071933214,0.011479695,0.6624913,0.31720218],"study_design_scores_gemma":[0.00018530845,0.0004955513,0.015475583,0.0030243548,0.00035118847,0.0018735777,0.0015831896,0.0044660727,0.002400121,0.12814267,0.8416051,0.0003973332],"about_ca_topic_score_codex":0.0017926439,"about_ca_topic_score_gemma":0.004261894,"teacher_disagreement_score":0.92854536,"about_ca_system_score_codex":0.0027465886,"about_ca_system_score_gemma":0.0042951913,"threshold_uncertainty_score":0.37789285},"labels":[],"label_agreement":null},{"id":"W3217598503","doi":"10.32920/ryerson.14645355.v1","title":"Microblog summarization based on sentiment and aspect analysis","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Wilfrid Laurier University; Toronto Metropolitan University; University of Waterloo","funders":"","keywords":"Automatic summarization; Microblogging; Sentiment analysis; Social media; Computer science; Information retrieval; Baseline (sea); Cluster analysis; Multi-document summarization; Annotation; Natural language processing; Data science; Artificial intelligence; World Wide Web","score_opus":0.009684912507483769,"score_gpt":0.263804563741241,"score_spread":0.25411965123375724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3217598503","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23870215,0.0022083344,0.73147196,0.0008962441,0.0005006314,0.0013315363,0.0070774886,0.009661342,0.008150413],"genre_scores_gemma":[0.4250402,0.0014330064,0.54698664,0.00015836112,0.0009456891,0.00070387445,0.015612593,0.00077276997,0.008346947],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99946004,0.00010482223,0.00007159553,0.0001094962,0.00019141313,0.00006259992],"domain_scores_gemma":[0.9978855,0.000507356,0.00030271927,0.00015940992,0.0010581348,0.00008685203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006930597,0.0011361658,0.0007982749,0.0044519263,0.0006023171,0.0013150509,0.00044450132,0.00037874424,0.001614424],"category_scores_gemma":[0.0031018977,0.00025651904,0.0006801919,0.0027339673,0.00018613116,0.0013495969,0.000694063,0.0005549684,0.0014884702],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000607295,0.00020175875,0.0071421512,0.00079994247,0.00019738352,0.00041616257,0.0013060757,0.007150688,0.14122717,0.0023016327,0.017940827,0.820709],"study_design_scores_gemma":[0.00015935613,0.0012597425,0.050518114,0.00019798496,0.00085694797,0.0010839642,0.0036612467,0.6009088,0.24294461,0.016769845,0.081438705,0.00020065582],"about_ca_topic_score_codex":0.0016736678,"about_ca_topic_score_gemma":0.0030687512,"teacher_disagreement_score":0.0044519263,"about_ca_system_score_codex":0.00033973585,"about_ca_system_score_gemma":0.00045703942,"threshold_uncertainty_score":0.005400777},"labels":[],"label_agreement":null},{"id":"W342307487","doi":"10.1007/978-3-319-13332-4_21","title":"A Comparison of Graph-Based and Statistical Metrics for Learning Domain Keywords","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Computer science; Pointwise mutual information; Graph; Pointwise; Artificial intelligence; Phrase; Machine learning; Data mining; Mutual information; Theoretical computer science; Mathematics","score_opus":0.021132709711514663,"score_gpt":0.3197664553226307,"score_spread":0.29863374561111605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W342307487","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14827523,0.016766481,0.81277466,0.0020257332,0.0005012693,0.0003237713,0.0043642484,0.0038239823,0.011144758],"genre_scores_gemma":[0.6727279,0.0041037304,0.3107551,0.0002566456,0.00030063,0.0002846913,0.007357826,0.00079649105,0.0034169652],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99326706,0.0026178507,0.00051738473,0.000796454,0.0025994573,0.00020175352],"domain_scores_gemma":[0.959436,0.02871728,0.00186873,0.003189042,0.0059003583,0.00088865886],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00665919,0.0010368351,0.0013824939,0.01070774,0.00067653303,0.0031624069,0.0016327866,0.001646034,0.0022905467],"category_scores_gemma":[0.035437796,0.00025237794,0.0007539647,0.008507063,0.0010814643,0.006519321,0.0022287725,0.0012185205,0.00083519815],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009920071,0.00032766987,0.030552227,0.0010289736,0.0005105113,0.00007678609,0.0002773244,0.067757666,0.0044502895,0.0418105,0.015899163,0.8363168],"study_design_scores_gemma":[0.00010125807,0.0012076863,0.022752581,0.00026119142,0.00026695302,0.00039954702,0.00052285986,0.8250442,0.0052856575,0.13000557,0.014039834,0.000112688234],"about_ca_topic_score_codex":0.003592203,"about_ca_topic_score_gemma":0.005029838,"teacher_disagreement_score":0.01070774,"about_ca_system_score_codex":0.0021011091,"about_ca_system_score_gemma":0.0014554621,"threshold_uncertainty_score":0.035217583},"labels":[],"label_agreement":null},{"id":"W366205791","doi":"10.1007/978-94-017-9112-0_34","title":"Understanding Design Concept Identification","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Identification (biology); Computer science; Brainstorming; Selection (genetic algorithm); Process (computing); Identifier; Workflow; Cognitive science; Artificial intelligence; Programming language; Psychology","score_opus":0.1505184642011906,"score_gpt":0.2905078990430491,"score_spread":0.13998943484185847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W366205791","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040129866,0.0036789535,0.749364,0.0023566517,0.0005012885,0.00015504229,0.0002814168,0.0013776991,0.23827203],"genre_scores_gemma":[0.103203095,0.00654066,0.51686364,0.0011249674,0.0003508217,0.00034967117,0.0020172603,0.0012220155,0.36832783],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99895513,0.00037293223,0.0000569183,0.0002162656,0.00034614277,0.000052586733],"domain_scores_gemma":[0.9987029,0.0007556674,0.000047653048,0.0001894014,0.00027763788,0.000026760219],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001131786,0.0013776447,0.0005622161,0.0023378169,0.0013216314,0.0050556213,0.0010883388,0.0011977501,0.027803188],"category_scores_gemma":[0.003964563,0.00066818635,0.0007265793,0.0014305856,0.002274892,0.007719162,0.0018172995,0.0028551994,0.01107616],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019923526,0.000036992766,0.00023105532,0.00040594037,0.000012207496,0.00012679324,0.0039155125,0.0011890484,0.0021639057,0.5080456,0.04633092,0.43752205],"study_design_scores_gemma":[0.0000059926724,0.00001815717,0.00026727887,0.0005633797,0.000020399788,0.0003729662,0.0015764538,0.0070629087,0.00329449,0.37213913,0.6146583,0.000020591335],"about_ca_topic_score_codex":0.0016588389,"about_ca_topic_score_gemma":0.0023829262,"teacher_disagreement_score":0.027803188,"about_ca_system_score_codex":0.0019854268,"about_ca_system_score_gemma":0.001661167,"threshold_uncertainty_score":0.0930109},"labels":[],"label_agreement":null},{"id":"W36643226","doi":"10.1007/1-4020-3670-1_10","title":"A Cognitive Framework for Human Information Behavior: The Place of Metaphor in Human Information Organizing Behavior","year":2006,"lang":"en","type":"book-chapter","venue":"Information science and knowledge management","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Nexus (standard); Metaphor; Unconscious mind; Perspective (graphical); Cognitive science; Cognition; Psychology; Sociology; Social psychology; Computer science; Psychoanalysis; Artificial intelligence; Linguistics; Philosophy; Neuroscience","score_opus":0.0241227843392514,"score_gpt":0.31908846873074403,"score_spread":0.2949656843914926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W36643226","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023258777,0.020111846,0.6261291,0.028329628,0.0008067878,0.00013450964,0.00028273446,0.0004657797,0.30048084],"genre_scores_gemma":[0.71942097,0.008060847,0.25577718,0.0019988022,0.00044042178,0.00039842067,0.0001924409,0.00020522908,0.013505776],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989058,0.0006494609,0.000047258996,0.00013406537,0.00018980716,0.000073610165],"domain_scores_gemma":[0.9978015,0.0014670932,0.00014190625,0.00017985019,0.00022729665,0.00018239624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021909946,0.00090785587,0.0006244598,0.0020209076,0.0016228497,0.0069744415,0.0019621004,0.0021067294,0.0039567715],"category_scores_gemma":[0.004496006,0.00039880627,0.00074312725,0.0025370943,0.01660751,0.013203896,0.0017406688,0.0027680907,0.0005571184],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000066554476,0.000011404294,0.00011189055,0.00007639104,0.000007999275,0.00004529971,0.003356579,0.000559222,0.00022138994,0.9876286,0.001853153,0.006121398],"study_design_scores_gemma":[0.000006143612,0.00001189851,0.00016598697,0.00006424874,0.0000070865794,0.000119903736,0.0012481911,0.0022979733,0.00012138459,0.97840106,0.017545477,0.000010713618],"about_ca_topic_score_codex":0.0024872809,"about_ca_topic_score_gemma":0.0024358958,"teacher_disagreement_score":0.0069744415,"about_ca_system_score_codex":0.0022279287,"about_ca_system_score_gemma":0.0023209963,"threshold_uncertainty_score":0.01616484},"labels":[],"label_agreement":null},{"id":"W4210562377","doi":"10.1109/icmla52953.2021.00204","title":"Sentiment Analysis of StockTwits Using Transformer Models","year":2021,"lang":"en","type":"article","venue":"2021 20th IEEE International Conference on Machine Learning and Applications (ICMLA)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Transformer; Electrical engineering; Engineering; Voltage","score_opus":0.053578207070827055,"score_gpt":0.35676368059054697,"score_spread":0.3031854735197199,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210562377","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.880366,0.0008573694,0.103892446,0.00059696124,0.00027920527,0.00018803826,0.0031561754,0.0016636372,0.009000197],"genre_scores_gemma":[0.984463,0.00020844158,0.00972883,0.000056494242,0.000054344702,0.000034562803,0.0023331095,0.00003451475,0.0030867704],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99974626,0.000052004627,0.000023045573,0.000059466638,0.00006999282,0.000049245442],"domain_scores_gemma":[0.99964976,0.00012578476,0.000045864203,0.000025610716,0.00012935774,0.000023554736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006026731,0.0005931497,0.00031874442,0.0010586557,0.00015408148,0.00066002365,0.0002532643,0.00028973597,0.0016672211],"category_scores_gemma":[0.0015292037,0.00011238264,0.0006044354,0.0005462361,0.00015801679,0.00088527374,0.00035390304,0.0004655092,0.0010660473],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017826466,0.00077480625,0.13003345,0.00037863993,0.00033449783,0.0008870177,0.0007203148,0.1542451,0.058072686,0.0063017556,0.024454162,0.62201506],"study_design_scores_gemma":[0.000013271984,0.00009005638,0.013951367,0.0000087359385,0.00003013696,0.000054783108,0.00013365451,0.97893655,0.0040973756,0.0011540065,0.0015174547,0.000012632227],"about_ca_topic_score_codex":0.0062963367,"about_ca_topic_score_gemma":0.007441737,"teacher_disagreement_score":0.0062963367,"about_ca_system_score_codex":0.0004629086,"about_ca_system_score_gemma":0.0003259059,"threshold_uncertainty_score":0.01251936},"labels":[],"label_agreement":null},{"id":"W4211085863","doi":"10.1017/cbo9780511500107.002","title":"Preliminaries","year":2002,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Sketch; Foundation (evidence); Epistemology; Empirical research; Reading (process); Domain (mathematical analysis); Computer science; Term (time); Management science; Linguistics; Mathematics; Philosophy; Engineering; History; Algorithm","score_opus":0.022315069497845428,"score_gpt":0.20147244822639773,"score_spread":0.1791573787285523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4211085863","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049732146,0.019575205,0.14876254,0.015982216,0.0057740165,0.00059637096,0.010642809,0.0016118733,0.7920818],"genre_scores_gemma":[0.1042681,0.031277977,0.18622114,0.015780387,0.006028128,0.0025500578,0.036365155,0.0024736347,0.6150354],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972184,0.0008500096,0.00032016207,0.0007368315,0.0006351817,0.00023951469],"domain_scores_gemma":[0.99675804,0.0014265982,0.0002591589,0.00057222473,0.0007937784,0.00019023957],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002170137,0.0013302527,0.0010326235,0.0029862982,0.003858421,0.005951757,0.0029287557,0.0027024276,0.16697286],"category_scores_gemma":[0.009471377,0.0006043043,0.0010737139,0.0040631504,0.0025919543,0.0068348735,0.004195935,0.0037972538,0.08988324],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056352375,0.00006277395,0.00047535734,0.00047863304,0.000012249886,0.0002665364,0.0012903549,0.0004401887,0.00039440126,0.7473357,0.18068326,0.06850423],"study_design_scores_gemma":[0.0000049135633,0.00001568877,0.0001939269,0.0001440701,0.000004071918,0.0002550108,0.00025759905,0.0002455401,0.000114002316,0.11775415,0.8809998,0.000011095623],"about_ca_topic_score_codex":0.003962906,"about_ca_topic_score_gemma":0.0030014066,"teacher_disagreement_score":0.8330271,"about_ca_system_score_codex":0.0027365256,"about_ca_system_score_gemma":0.0023396248,"threshold_uncertainty_score":0.55857986},"labels":[],"label_agreement":null},{"id":"W4213159392","doi":"10.21203/rs.3.rs-1041491/v1","title":"A Proposed Method for Residual Citation Allocation Based on Citation Contexts’ Similarity","year":2021,"lang":"en","type":"preprint","venue":"Research Square","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Citation; Residual; Similarity (geometry); Computer science; Information retrieval; Data science; Artificial intelligence; Algorithm; World Wide Web","score_opus":0.11728681695612835,"score_gpt":0.47453324176773076,"score_spread":0.3572464248116024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4213159392","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08054637,0.0009821767,0.9123437,0.00024553767,0.00016981624,0.00049073354,0.00037705142,0.0012970807,0.0035476096],"genre_scores_gemma":[0.42785975,0.00037733605,0.5671024,0.00005897954,0.00022422046,0.00059287925,0.0005112592,0.00016605677,0.0031071443],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99521387,0.0013009137,0.00058660214,0.0011147081,0.0015722305,0.00021174787],"domain_scores_gemma":[0.99057406,0.0036982878,0.0009911985,0.0010876635,0.0033497405,0.00029906238],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0045792754,0.00083697867,0.0010887878,0.011741357,0.0010948286,0.0021086973,0.0015338865,0.00073875865,0.0028759309],"category_scores_gemma":[0.022977382,0.00030561074,0.00089372875,0.007055924,0.0006081356,0.0023504605,0.0018458177,0.00073663105,0.0009913372],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025340897,0.00017888888,0.0129018575,0.0003034019,0.00015714577,0.00010598592,0.0005879342,0.01276375,0.009622054,0.012922009,0.0019114426,0.9482922],"study_design_scores_gemma":[0.00024798582,0.00061768125,0.04132658,0.00021722485,0.0007665922,0.00075330475,0.0011788484,0.8146027,0.029310385,0.07911701,0.031623304,0.0002385423],"about_ca_topic_score_codex":0.002448725,"about_ca_topic_score_gemma":0.0025242574,"teacher_disagreement_score":0.99542075,"about_ca_system_score_codex":0.0008803995,"about_ca_system_score_gemma":0.0022053006,"threshold_uncertainty_score":0.024217844},"labels":[],"label_agreement":null},{"id":"W4220686815","doi":"10.29173/cais1293","title":"Correlation of term usage and term indexing frequencies","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"Western University","funders":"","keywords":"Zipf's law; Term (time); Search engine indexing; Information retrieval; Rank (graph theory); Computer science; Cluster analysis; Index (typography); Plot (graphics); Word (group theory); Data mining; Statistics; Mathematics; Artificial intelligence; World Wide Web","score_opus":0.01800631411748159,"score_gpt":0.24861207457086676,"score_spread":0.23060576045338516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220686815","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97267026,0.0020387077,0.006808303,0.00030016145,0.00010862909,0.00011237209,0.0065103215,0.00031273946,0.011138459],"genre_scores_gemma":[0.9865428,0.00070681644,0.0034777788,0.00003837498,0.00007893252,0.00012240998,0.005882671,0.00015090567,0.0029994196],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9845137,0.0033045814,0.0026797235,0.0017949974,0.0069404007,0.0007665999],"domain_scores_gemma":[0.79933095,0.13633022,0.03037205,0.009399674,0.022804158,0.0017629483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077301343,0.0002777051,0.0007103941,0.010260264,0.00048065928,0.0026287357,0.0007767488,0.00043665603,0.0050181584],"category_scores_gemma":[0.09990515,0.0003251471,0.0007097749,0.019027736,0.0008616393,0.0021718286,0.0012377634,0.0009944765,0.0019774004],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002767644,0.00005162576,0.9641506,0.00026022777,0.00022451764,0.0001345055,0.0011563397,0.00066992256,0.00095374417,0.0008602903,0.0015803239,0.029681105],"study_design_scores_gemma":[0.0000061322876,0.000098937264,0.99085534,0.00004537383,0.000060898517,0.0004309009,0.0007973971,0.002029338,0.0008403646,0.00067991094,0.004119655,0.000035758832],"about_ca_topic_score_codex":0.007966128,"about_ca_topic_score_gemma":0.0074675363,"teacher_disagreement_score":0.010260264,"about_ca_system_score_codex":0.0009767895,"about_ca_system_score_gemma":0.0008983715,"threshold_uncertainty_score":0.040881395},"labels":[],"label_agreement":null},{"id":"W4220849857","doi":"10.29173/cais1282","title":"IdeaMap: A sophisticated graphical idea processor","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science","score_opus":0.01780833033444457,"score_gpt":0.2552334052985238,"score_spread":0.23742507496407922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220849857","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012071044,0.0004647737,0.7223173,0.00027508178,0.0003101489,0.00019565044,0.0063593714,0.2544669,0.014403726],"genre_scores_gemma":[0.04939586,0.0010989571,0.7643655,0.0010908995,0.0002907534,0.0011631107,0.027549453,0.08619247,0.068853006],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990375,0.00017108938,0.000092168964,0.0001637661,0.00044723856,0.00008829617],"domain_scores_gemma":[0.9971323,0.0013668106,0.00009011912,0.00061537104,0.00062296214,0.00017246125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014186287,0.002337748,0.0009726311,0.0020421387,0.0004973315,0.0031262066,0.00360381,0.0017018433,0.21758686],"category_scores_gemma":[0.006798356,0.0015642837,0.0014689735,0.0024801842,0.0006558316,0.005899883,0.0031246347,0.0023908333,0.090607315],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014438266,0.00013410188,0.0005581714,0.0010447019,0.00011363194,0.0002726921,0.00022055412,0.0049206675,0.013498243,0.020439142,0.4569708,0.5003835],"study_design_scores_gemma":[0.0014769087,0.00039652188,0.0008212043,0.00035104476,0.00017867223,0.00096107816,0.00017397558,0.13715512,0.056742955,0.10018041,0.7012645,0.0002976847],"about_ca_topic_score_codex":0.0012533129,"about_ca_topic_score_gemma":0.0013294606,"teacher_disagreement_score":0.21758686,"about_ca_system_score_codex":0.00048458664,"about_ca_system_score_gemma":0.00083701266,"threshold_uncertainty_score":0.72790056},"labels":[],"label_agreement":null},{"id":"W4220959518","doi":"10.5281/zenodo.6366635","title":"Identifying science in the news: An assessment of the precision and recall of Altmetric.com news mention data","year":2022,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; Simon Fraser University","funders":"","keywords":"Recall; Computer science; Altmetrics; Fake news; Data science; Information retrieval; Psychology; Internet privacy; Cognitive psychology","score_opus":0.09913731076647833,"score_gpt":0.3735945198395886,"score_spread":0.2744572090731102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220959518","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8904936,0.009576524,0.052221335,0.0016771927,0.000751423,0.0016797475,0.015409556,0.0025072433,0.025683384],"genre_scores_gemma":[0.88225645,0.0018039638,0.09361181,0.0007330422,0.00054514094,0.0014301224,0.015746927,0.00050690275,0.0033656391],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9008672,0.0384018,0.018720828,0.011252213,0.029441543,0.001316369],"domain_scores_gemma":[0.45711288,0.38392273,0.056270953,0.035100352,0.0657886,0.0018045589],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.1197923,0.0010943761,0.0018645169,0.03666399,0.0015645273,0.0063555073,0.0022737721,0.0023829534,0.0014103289],"category_scores_gemma":[0.33409074,0.00075771735,0.0021364987,0.018966269,0.0018329796,0.0062208734,0.0044538756,0.0011659686,0.0017317027],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014333456,0.0004546942,0.7326017,0.0046623917,0.003204586,0.00057108636,0.01497906,0.0025780015,0.006198835,0.0025151595,0.015581615,0.2152195],"study_design_scores_gemma":[0.00023483255,0.0010477557,0.82895976,0.0023956257,0.003408558,0.002577587,0.011236176,0.038146157,0.02725935,0.0057944655,0.078266144,0.000673551],"about_ca_topic_score_codex":0.0032360156,"about_ca_topic_score_gemma":0.004023233,"teacher_disagreement_score":0.963336,"about_ca_system_score_codex":0.00128156,"about_ca_system_score_gemma":0.0009947219,"threshold_uncertainty_score":0.6335298},"labels":[],"label_agreement":null},{"id":"W4221028329","doi":"10.1037/cep0000277","title":"Beauty and truth, truth and beauty: Chiastic structure increases the subjective accuracy of statements.","year":2022,"lang":"en","type":"article","venue":"Canadian Journal of Experimental Psychology/Revue canadienne de psychologie expérimentale","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Beauty; Happiness; Fluency; Psychology; PsycINFO; Cognitive psychology; Computer science; Social psychology; Linguistics; Aesthetics; Philosophy; MEDLINE; Mathematics education; Law","score_opus":0.022957224937724587,"score_gpt":0.3252512354830567,"score_spread":0.3022940105453321,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4221028329","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92713004,0.0016929484,0.0069765165,0.0024174075,0.0002594065,0.00007261264,0.0009503755,0.000109379784,0.060391214],"genre_scores_gemma":[0.9935568,0.0003396646,0.0035365168,0.00019904901,0.00007156702,0.000029937088,0.00022785766,0.000016888647,0.0020216585],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9989574,0.00043919508,0.0001033729,0.0001577948,0.0003115818,0.000030684936],"domain_scores_gemma":[0.9730195,0.0163214,0.007921326,0.0013079023,0.00087482674,0.0005551491],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014921152,0.00016918169,0.00012359324,0.00053235027,0.00037663357,0.0019699184,0.0001846331,0.0005963879,0.013002647],"category_scores_gemma":[0.033491917,0.00013018666,0.00017456364,0.0005989924,0.00085625675,0.0018042225,0.0007544089,0.0005551171,0.00071346725],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0063911304,0.0006813754,0.28069687,0.0027866596,0.0003819717,0.00094708113,0.01834644,0.0019278986,0.11266026,0.06837659,0.017146308,0.4896574],"study_design_scores_gemma":[0.00014345351,0.0011940995,0.9030131,0.00045542305,0.0003384715,0.0013903029,0.0040201293,0.005260539,0.020324826,0.039900225,0.023870949,0.00008850615],"about_ca_topic_score_codex":0.00068753073,"about_ca_topic_score_gemma":0.0020129497,"teacher_disagreement_score":0.013002647,"about_ca_system_score_codex":0.0005015894,"about_ca_system_score_gemma":0.00027225347,"threshold_uncertainty_score":0.04349816},"labels":[],"label_agreement":null},{"id":"W4229010387","doi":"10.16995/dscn.8106","title":"Concept Detection in Philosophical Corpora","year":2022,"lang":"en","type":"article","venue":"Digital Studies / Le champ numérique","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Cosine similarity; Natural language processing; Document retrieval; Information retrieval; Similarity (geometry); Artificial intelligence; Word (group theory); Embedding; Linguistics; Pattern recognition (psychology); Philosophy; Image (mathematics)","score_opus":0.025000278519651958,"score_gpt":0.2731868783958897,"score_spread":0.24818659987623773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4229010387","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22231236,0.0074427156,0.669117,0.0032356621,0.0012029083,0.0030456795,0.021282034,0.0065730326,0.065788634],"genre_scores_gemma":[0.2980245,0.0018770803,0.66138613,0.00042528083,0.00026062754,0.0020627985,0.025181329,0.0007217659,0.010060548],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99063724,0.003558018,0.0010965075,0.0019387252,0.0025483079,0.00022120749],"domain_scores_gemma":[0.9696107,0.018177615,0.0018072632,0.0041000056,0.0058715707,0.00043278368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069454196,0.00093800743,0.000809894,0.016450567,0.0020149671,0.0046672244,0.001880023,0.0011070141,0.014005325],"category_scores_gemma":[0.05294692,0.00057363877,0.001052524,0.01243466,0.0018119066,0.0066771903,0.0044355234,0.0014565297,0.0051861443],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059767545,0.00029092783,0.011548905,0.0046497653,0.0001829334,0.0011649361,0.014418588,0.004145595,0.03088044,0.10184624,0.03442063,0.7958534],"study_design_scores_gemma":[0.0003067319,0.00052352925,0.030744782,0.0017158401,0.0002923673,0.004019836,0.024783934,0.08464414,0.06456034,0.13620667,0.6518489,0.0003529525],"about_ca_topic_score_codex":0.002452279,"about_ca_topic_score_gemma":0.002656937,"teacher_disagreement_score":0.016450567,"about_ca_system_score_codex":0.002090174,"about_ca_system_score_gemma":0.0025448473,"threshold_uncertainty_score":0.04685253},"labels":[],"label_agreement":null},{"id":"W4229813582","doi":"10.1017/cbo9781139979573.006","title":"Path analysis and maximum likelihood","year":2016,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Personality psychology; Discipline; Epistemology; Path (computing); Sociology; History; Psychology; Social science; Social psychology; Computer science; Philosophy; Personality","score_opus":0.01060950749121917,"score_gpt":0.2000616385044859,"score_spread":0.18945213101326674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4229813582","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0024641978,0.004426762,0.9869265,0.0017596683,0.00018606301,0.00017085896,0.00054961635,0.00068129157,0.00283493],"genre_scores_gemma":[0.117970005,0.00645855,0.86398184,0.00053304236,0.00046848913,0.0016642403,0.0019713384,0.0008269381,0.006125473],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98014176,0.01674711,0.0004898498,0.0014997146,0.00075481395,0.00036674953],"domain_scores_gemma":[0.9417039,0.053654645,0.0013893099,0.0017350002,0.0011841245,0.00033300233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017831318,0.0030989351,0.0045050727,0.005604549,0.0020066195,0.0043889224,0.0036159789,0.00338045,0.018985491],"category_scores_gemma":[0.073694184,0.0020868955,0.003544391,0.009360819,0.003914143,0.0056187203,0.0042894073,0.007134989,0.0049822177],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003001547,0.00023113514,0.0067538545,0.0013949461,0.0020836892,0.00041064995,0.0009630246,0.23099118,0.00021472762,0.46049848,0.023869026,0.27228913],"study_design_scores_gemma":[0.00006189049,0.000071743576,0.0009657252,0.00023419925,0.00010515926,0.0001236943,0.00025549985,0.32620487,0.00015789828,0.6580047,0.013734628,0.00007993947],"about_ca_topic_score_codex":0.013225479,"about_ca_topic_score_gemma":0.009697526,"teacher_disagreement_score":0.018985491,"about_ca_system_score_codex":0.0026553744,"about_ca_system_score_gemma":0.0045585125,"threshold_uncertainty_score":0.09430212},"labels":[],"label_agreement":null},{"id":"W423015428","doi":"10.1016/j.joi.2015.05.001","title":"Modelling count response variables in informetric studies: Comparison among count, linear, and lognormal regression models","year":2015,"lang":"en","type":"article","venue":"Journal of Informetrics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Capital Medical University","keywords":"Count data; Statistics; Negative binomial distribution; Poisson regression; Akaike information criterion; Mathematics; Regression analysis; Linear regression; Poisson distribution; Overdispersion; Log-normal distribution; Generalized linear model; Binomial regression; Regression diagnostic; Econometrics; Polynomial regression; Population","score_opus":0.103223804081119,"score_gpt":0.3570945740796192,"score_spread":0.2538707699985002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W423015428","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27917996,0.005025704,0.7071362,0.0030022059,0.00031492038,0.000931716,0.0010856166,0.00044387588,0.0028797411],"genre_scores_gemma":[0.8191155,0.0038136055,0.16942944,0.0005021491,0.00027053748,0.0016875102,0.0010815185,0.00022114595,0.0038784887],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.93975127,0.05364316,0.0016212679,0.0022376846,0.0021183544,0.0006283176],"domain_scores_gemma":[0.454904,0.5239224,0.010077566,0.0051069357,0.005247912,0.0007411545],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.070555255,0.0017706831,0.00243211,0.004490251,0.0008447333,0.005542858,0.005063356,0.0037855976,0.005513296],"category_scores_gemma":[0.27881673,0.00085572124,0.0035437006,0.0064615053,0.0021470515,0.008207985,0.0026407635,0.0035166447,0.001320098],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0074979346,0.0015120083,0.1572095,0.0049407305,0.0041814153,0.00083086453,0.005764341,0.40571794,0.0010913606,0.13484067,0.004204678,0.27220848],"study_design_scores_gemma":[0.00032645126,0.0010868552,0.01134864,0.00069594657,0.0011861789,0.00034665718,0.0018874602,0.90548056,0.0006569601,0.07434648,0.0024964663,0.00014137963],"about_ca_topic_score_codex":0.009292494,"about_ca_topic_score_gemma":0.0077116527,"teacher_disagreement_score":0.99550974,"about_ca_system_score_codex":0.0030722108,"about_ca_system_score_gemma":0.0030965116,"threshold_uncertainty_score":0.37313628},"labels":[{"model":"gemma","categories":["metaresearch"],"domain":"methods","study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":["bibliometrics"],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W4231486013","doi":"10.1007/978-1-4614-6170-8_110113","title":"Automatic Document Topic Identification","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Identification (biology); Computer science; Information retrieval; Natural language processing; Biology","score_opus":0.010640889070436843,"score_gpt":0.2615044887871174,"score_spread":0.25086359971668054,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231486013","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014524498,0.01382431,0.8836918,0.000863069,0.0017342356,0.00024917742,0.0031050788,0.01711548,0.06489226],"genre_scores_gemma":[0.108049914,0.009842523,0.6419629,0.0003528025,0.001314695,0.0003065397,0.016863931,0.0028206697,0.21848601],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994722,0.00007566993,0.000035603643,0.00015406396,0.00020945608,0.000053000826],"domain_scores_gemma":[0.99906737,0.0003082166,0.000051564395,0.00015660048,0.0003721793,0.00004405273],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00059058965,0.0010263909,0.0007905196,0.0037944196,0.0007927374,0.002141223,0.0008301325,0.0006677974,0.016620964],"category_scores_gemma":[0.0016103988,0.00044115575,0.00076782814,0.0027629514,0.00035307067,0.002075815,0.0011887657,0.0010172767,0.021033082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008036694,0.00004104953,0.00033840263,0.00033075557,0.000024131887,0.00008188352,0.00013137038,0.0007375085,0.031659994,0.00492744,0.06791469,0.8937325],"study_design_scores_gemma":[0.00003827506,0.00014165482,0.005541773,0.00035842363,0.00019628262,0.0024643587,0.0004957386,0.08545154,0.16059856,0.030161003,0.71443236,0.00011999545],"about_ca_topic_score_codex":0.00069918286,"about_ca_topic_score_gemma":0.0010818715,"teacher_disagreement_score":0.016620964,"about_ca_system_score_codex":0.00046479638,"about_ca_system_score_gemma":0.00059551036,"threshold_uncertainty_score":0.05560267},"labels":[],"label_agreement":null},{"id":"W4231667695","doi":"10.2196/preprints.16891","title":"Using the Kano Model to Display the Association between Percentages of Keywords within Abstracts and Article Citations: A Bibliometric Study for JMIR Journals (Preprint)","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Citation; Association (psychology); Library science; Odds; Medicine; Psychology; Computer science; Logistic regression","score_opus":0.10961414235828783,"score_gpt":0.4221488472155437,"score_spread":0.3125347048572559,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231667695","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88705146,0.0022918493,0.07812606,0.0023536396,0.0003091174,0.0014531334,0.014673517,0.002017068,0.011724145],"genre_scores_gemma":[0.9702598,0.0003771941,0.022498958,0.00012586435,0.000096827585,0.001270702,0.0036858947,0.00026262045,0.0014222206],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9624527,0.01989558,0.0050503192,0.0045078564,0.006908978,0.0011845059],"domain_scores_gemma":[0.75251645,0.21078552,0.016563565,0.0080969045,0.010929873,0.0011076289],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.04567155,0.0011710391,0.001947226,0.033037663,0.0012161151,0.0069115646,0.0020546077,0.001557981,0.006269926],"category_scores_gemma":[0.20982221,0.0006841035,0.0050901426,0.047764923,0.0017488418,0.008527821,0.0033050396,0.0017705264,0.0021331902],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010486074,0.0003813294,0.8612693,0.0021542658,0.0045098322,0.0004145778,0.01216873,0.01006188,0.0010757269,0.00789374,0.010405378,0.08861652],"study_design_scores_gemma":[0.0002365563,0.001027417,0.72051066,0.00072995154,0.002334689,0.0009836464,0.02031067,0.2111471,0.002262774,0.024665415,0.015244632,0.0005465023],"about_ca_topic_score_codex":0.00944193,"about_ca_topic_score_gemma":0.0056740604,"teacher_disagreement_score":0.96696234,"about_ca_system_score_codex":0.0024559302,"about_ca_system_score_gemma":0.0026064324,"threshold_uncertainty_score":0.24153715},"labels":[],"label_agreement":null},{"id":"W4233701208","doi":"10.1007/978-1-4939-7131-2_101352","title":"Time-Sensitive Recommendation","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science","score_opus":0.01645824349576573,"score_gpt":0.2625762804319835,"score_spread":0.24611803693621775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4233701208","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00869874,0.0117347175,0.83896375,0.0023233632,0.00385351,0.0003297685,0.003036943,0.0076558474,0.1234033],"genre_scores_gemma":[0.18308334,0.013810763,0.3798839,0.0015451503,0.00384358,0.0002605159,0.008194065,0.0017192658,0.40765935],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99830186,0.00021863257,0.00007479844,0.0004291556,0.000850605,0.00012486283],"domain_scores_gemma":[0.9976866,0.0006272002,0.00008826048,0.0009702961,0.0005366357,0.000090996626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011074007,0.0015714795,0.0015113077,0.0015932324,0.00075552723,0.0022622892,0.0018875191,0.0014032178,0.030064253],"category_scores_gemma":[0.0050929952,0.0005940535,0.00083754474,0.0033531934,0.0004069027,0.0039933627,0.0010887054,0.002304976,0.025294922],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023178417,0.00019563094,0.0002814811,0.00030120942,0.00010565613,0.000068575144,0.00004557029,0.013763839,0.008374592,0.04515951,0.17976215,0.75170994],"study_design_scores_gemma":[0.000075163895,0.00024210711,0.0013835554,0.00016910653,0.00020550167,0.001009359,0.000091568945,0.46048024,0.022639893,0.21075918,0.3027899,0.00015443178],"about_ca_topic_score_codex":0.0027103338,"about_ca_topic_score_gemma":0.0038133108,"teacher_disagreement_score":0.030064253,"about_ca_system_score_codex":0.0010482061,"about_ca_system_score_gemma":0.0008757163,"threshold_uncertainty_score":0.10057497},"labels":[],"label_agreement":null},{"id":"W4235659173","doi":"10.4018/9781605661728.ch009.ch000","title":"A Model for Estimating the Savings from Dimensional vs. Keyword Search","year":2011,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Metadata; Computer science; Zipf's law; Process (computing); Keyword search; Information retrieval; Search cost; Search engine; Data mining; World Wide Web; Economics; Mathematics; Statistics; Microeconomics","score_opus":0.04719463325982439,"score_gpt":0.2919141828493511,"score_spread":0.2447195495895267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4235659173","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1173232,0.001175133,0.8482863,0.0024123883,0.000059259346,0.00041016564,0.0020659848,0.0011300185,0.027137663],"genre_scores_gemma":[0.71049756,0.001463949,0.26490664,0.00039089774,0.000067532856,0.0009901966,0.0017291073,0.00045310086,0.019501101],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99770397,0.0009669763,0.00013259596,0.0004176543,0.0005604451,0.00021835393],"domain_scores_gemma":[0.9764631,0.01961583,0.0015904407,0.0010309158,0.0010658144,0.00023390508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004341485,0.00093629083,0.0012569771,0.0022633625,0.000606247,0.003731717,0.0027396544,0.0026241664,0.011857048],"category_scores_gemma":[0.030080257,0.00089769723,0.0012035639,0.0042387503,0.0013555565,0.008087068,0.0012761251,0.0017738364,0.0032574912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028990785,0.00013731273,0.0047864346,0.00025022193,0.00008122099,0.0001568216,0.00024493702,0.7870596,0.0015542196,0.14303207,0.004111129,0.058296114],"study_design_scores_gemma":[0.000021501677,0.00007023124,0.0010488718,0.000028237635,0.000035453617,0.0001477258,0.000081334736,0.94750416,0.0004099114,0.048935853,0.0016819674,0.00003469845],"about_ca_topic_score_codex":0.008655791,"about_ca_topic_score_gemma":0.0059432522,"teacher_disagreement_score":0.011857048,"about_ca_system_score_codex":0.0040145773,"about_ca_system_score_gemma":0.0018170443,"threshold_uncertainty_score":0.03966582},"labels":[],"label_agreement":null},{"id":"W4236185729","doi":"10.1007/978-1-4614-6170-8_100575","title":"Time-Sensitive Recommendation","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science","score_opus":0.01281652761943931,"score_gpt":0.2481755232703999,"score_spread":0.2353589956509606,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4236185729","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008507945,0.012782454,0.8339907,0.002021615,0.003198075,0.0003045413,0.002550253,0.006802391,0.1298421],"genre_scores_gemma":[0.1717905,0.014384887,0.37097234,0.0012919172,0.0030817091,0.00022683693,0.006817934,0.0014271077,0.4300067],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984243,0.00019628821,0.000067555724,0.00037810658,0.00082033215,0.00011342796],"domain_scores_gemma":[0.99816316,0.00050906866,0.00007465094,0.0007341356,0.0004460835,0.00007291226],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010119071,0.0014728451,0.0014373056,0.0015863441,0.00072784367,0.0020557684,0.0018179683,0.001269206,0.027936552],"category_scores_gemma":[0.004119216,0.00057699904,0.0007559166,0.0033740252,0.0003989154,0.0036003636,0.0009812901,0.0020604578,0.022707142],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019390848,0.00017160295,0.00026223183,0.00026047713,0.000091120004,0.000062327235,0.000043627973,0.012707022,0.007839533,0.041907046,0.1596022,0.77685887],"study_design_scores_gemma":[0.00007116035,0.00024062689,0.0015538464,0.00017226975,0.00019261804,0.0011026504,0.000095495445,0.45197096,0.024561921,0.20099089,0.31888592,0.00016169746],"about_ca_topic_score_codex":0.0029087937,"about_ca_topic_score_gemma":0.0041319584,"teacher_disagreement_score":0.027936552,"about_ca_system_score_codex":0.0010351025,"about_ca_system_score_gemma":0.000789281,"threshold_uncertainty_score":0.09345704},"labels":[],"label_agreement":null},{"id":"W4236530352","doi":"10.32920/ryerson.14649648","title":"Fuzzy Thesauri Recommendation System For Web 2.0 Social networks","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Information retrieval; World Wide Web; The Internet; Scalability; Set (abstract data type); Node (physics); Social network (sociolinguistics); Recommender system; Order (exchange); Database; Social media","score_opus":0.026919574988326124,"score_gpt":0.30151589840289234,"score_spread":0.2745963234145662,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4236530352","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24606709,0.0035536531,0.6984055,0.0013102315,0.0004804528,0.0014825914,0.0033890428,0.020503664,0.024807787],"genre_scores_gemma":[0.72590476,0.0012022608,0.24851313,0.00024090582,0.000121485245,0.00041412367,0.0022910805,0.00011011984,0.021201978],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948007,0.000076448974,0.000058233018,0.00012816745,0.00020741823,0.000049658564],"domain_scores_gemma":[0.9995382,0.000085774474,0.00003533854,0.0000530206,0.0002546419,0.000033025233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071749487,0.00062438817,0.00098458,0.0018768954,0.0009912295,0.0010421263,0.0011365943,0.0011140615,0.005861525],"category_scores_gemma":[0.001658992,0.00026077055,0.00084305473,0.0010971663,0.00019312662,0.0010306421,0.00040275176,0.0005550573,0.0024740677],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023568005,0.0011008905,0.008562291,0.00067976065,0.00038207797,0.0010685807,0.00040790113,0.065736935,0.040917303,0.0060545946,0.031941094,0.84079176],"study_design_scores_gemma":[0.00012870177,0.0002455945,0.003705674,0.00003086843,0.00014602121,0.00025377065,0.00013189665,0.9733208,0.012252106,0.0018203134,0.007899892,0.00006442048],"about_ca_topic_score_codex":0.046441548,"about_ca_topic_score_gemma":0.055436008,"teacher_disagreement_score":0.046441548,"about_ca_system_score_codex":0.0012370893,"about_ca_system_score_gemma":0.0009285171,"threshold_uncertainty_score":0.092342496},"labels":[],"label_agreement":null},{"id":"W4238322342","doi":"10.31234/osf.io/svmtd","title":"The Semantic Librarian: A Search Engine Built from Vector-Space Models of Semantics","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; University of Winnipeg","funders":"","keywords":"Semantics (computer science); Space (punctuation); Computer science; Cognition; Semantic space; Information retrieval; World Wide Web; Cognitive science; Artificial intelligence; Psychology; Programming language","score_opus":0.031083942001109043,"score_gpt":0.281068873984278,"score_spread":0.24998493198316896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4238322342","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008329655,0.0029090883,0.7067997,0.0019853683,0.00029699266,0.0005798448,0.047396433,0.19517173,0.036531176],"genre_scores_gemma":[0.09766644,0.004482662,0.7462303,0.0011962649,0.00025263595,0.00091703393,0.09925803,0.019447656,0.03054905],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99917114,0.00019909772,0.00011230927,0.00015234144,0.0003121684,0.00005290982],"domain_scores_gemma":[0.9975605,0.0012371115,0.00014989356,0.000528224,0.0003593401,0.00016499538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018288776,0.0011713828,0.0011521933,0.010441844,0.00079501787,0.0037278149,0.0019833955,0.0013268605,0.030774014],"category_scores_gemma":[0.010942348,0.00058727985,0.0010724687,0.009671646,0.00067538215,0.0094548585,0.0036871205,0.0010134411,0.018538004],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006218436,0.00017938095,0.0023837087,0.0019515939,0.0003082305,0.00038621164,0.0007598536,0.0087787,0.0038074472,0.19129217,0.34526008,0.4442708],"study_design_scores_gemma":[0.00025938902,0.00013480599,0.0010064638,0.00034512568,0.00013588296,0.0006100222,0.00039921337,0.17570552,0.008110891,0.26172182,0.5514202,0.00015075051],"about_ca_topic_score_codex":0.0034757277,"about_ca_topic_score_gemma":0.0067149634,"teacher_disagreement_score":0.030774014,"about_ca_system_score_codex":0.0010473531,"about_ca_system_score_gemma":0.0022047427,"threshold_uncertainty_score":0.10294938},"labels":[],"label_agreement":null},{"id":"W4239379014","doi":"10.4018/9781591405573.ch105","title":"Hierarchical Document Clustering","year":2011,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Document clustering; Cluster analysis; Hierarchy; Hierarchical clustering; Computer science; Directory; Information retrieval; Tree (set theory); Subject (documents); Similarity (geometry); Complete-linkage clustering; Cluster (spacecraft); Fuzzy clustering; Artificial intelligence; World Wide Web; Canopy clustering algorithm; Mathematics; Combinatorics","score_opus":0.02025228869145076,"score_gpt":0.26863962020140814,"score_spread":0.24838733150995737,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4239379014","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0064507523,0.01234535,0.8938096,0.0010691832,0.0006660655,0.0017002213,0.022598097,0.015590641,0.04577007],"genre_scores_gemma":[0.045054287,0.006133341,0.8722796,0.00039702357,0.00036653655,0.0007164252,0.038031433,0.0014570289,0.03556426],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9960134,0.0007866863,0.00032016885,0.0011024436,0.0015646842,0.0002125882],"domain_scores_gemma":[0.99615854,0.001049774,0.00023908814,0.00077864,0.0016310499,0.00014299237],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022372738,0.0016095641,0.0016905103,0.011972126,0.0019823122,0.0048186807,0.0029508097,0.0013355255,0.02312193],"category_scores_gemma":[0.008381198,0.00059189956,0.0018651241,0.016140237,0.00070942054,0.0031199341,0.0020627023,0.001404138,0.03212331],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011132876,0.00011306555,0.0015338437,0.0015654251,0.00024324856,0.00017955997,0.0005347059,0.0071507175,0.0045084883,0.022892587,0.15682657,0.8043404],"study_design_scores_gemma":[0.0000763074,0.00013580776,0.004798617,0.0007735356,0.00035041466,0.0012786381,0.0008159422,0.12084069,0.013695309,0.08762411,0.76942146,0.00018918134],"about_ca_topic_score_codex":0.007145505,"about_ca_topic_score_gemma":0.011028554,"teacher_disagreement_score":0.02312193,"about_ca_system_score_codex":0.0018297553,"about_ca_system_score_gemma":0.0035795968,"threshold_uncertainty_score":0.07735056},"labels":[],"label_agreement":null},{"id":"W4239534043","doi":"10.1007/978-1-4939-7131-2_101211","title":"Social Trends Discovery","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Data science; Computer science","score_opus":0.023970749127788078,"score_gpt":0.2956227825873088,"score_spread":0.2716520334595207,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4239534043","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007978139,0.018427828,0.49932596,0.009757042,0.0040398985,0.0007480761,0.008827257,0.008014416,0.44288144],"genre_scores_gemma":[0.05863901,0.020145014,0.34373358,0.0013507776,0.003322382,0.0005539448,0.01900526,0.002156602,0.5510934],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999116,0.00012590634,0.00004352908,0.00021475926,0.00045180303,0.000048032827],"domain_scores_gemma":[0.99854666,0.00062191696,0.00007542105,0.00028687835,0.00037586474,0.00009310068],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011813585,0.0011676485,0.0007882524,0.0065171034,0.0015823338,0.0043732356,0.0014203842,0.0010650296,0.03905009],"category_scores_gemma":[0.004459539,0.00056568853,0.0011726477,0.0059993654,0.00064754725,0.0051856916,0.0024934297,0.0016188757,0.031984992],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034167584,0.00004795956,0.0005855187,0.0004210345,0.000052724838,0.00010929582,0.0003572316,0.0009639437,0.002060126,0.100352295,0.19515534,0.69986033],"study_design_scores_gemma":[0.000008757792,0.000022016142,0.0009983997,0.00024820084,0.000049829796,0.00050418836,0.00031593317,0.013487705,0.0045811753,0.16715346,0.8125968,0.00003359815],"about_ca_topic_score_codex":0.0014728458,"about_ca_topic_score_gemma":0.0028792007,"teacher_disagreement_score":0.03905009,"about_ca_system_score_codex":0.0012581317,"about_ca_system_score_gemma":0.0014311002,"threshold_uncertainty_score":0.13063562},"labels":[],"label_agreement":null},{"id":"W4242274911","doi":"10.1002/asi.1101","title":"User preferences in the classification of electronic bookmarks: Implications for a shared system","year":2001,"lang":"en","type":"article","venue":"Journal of the American Society for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Categorization; Computer science; Context (archaeology); The Internet; World Wide Web; Documentation; Sample (material); Information retrieval; Artificial intelligence","score_opus":0.018552468412853438,"score_gpt":0.3082505032618173,"score_spread":0.28969803484896384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4242274911","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9953591,0.000019348097,0.0020595805,0.00021916836,0.000003350397,0.00002646065,0.000011817411,0.000016642303,0.0022845448],"genre_scores_gemma":[0.99828774,0.000010273057,0.0013784253,0.000028791288,0.0000032235964,0.000014046642,0.000011574668,0.000004997444,0.0002610227],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9914539,0.0057342947,0.000431437,0.00059297984,0.001361776,0.0004256853],"domain_scores_gemma":[0.9582817,0.029837368,0.003350064,0.0028434417,0.0037479382,0.0019394433],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008959019,0.00016769783,0.0002518698,0.0008186895,0.0013678995,0.0041750707,0.0005224035,0.00076237705,0.0029320174],"category_scores_gemma":[0.039925378,0.00020596046,0.00031895324,0.0007225096,0.0012790912,0.0037928787,0.0014513922,0.00062631385,0.0003762715],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012405704,0.0011178628,0.6854614,0.00018936736,0.000082005645,0.0006433804,0.13694102,0.0012516544,0.015688391,0.0065841293,0.0013139116,0.14948624],"study_design_scores_gemma":[0.00010023656,0.0019588447,0.64714086,0.00019019145,0.00008876065,0.0012404086,0.2980502,0.022732282,0.007840058,0.010598204,0.009863597,0.0001963956],"about_ca_topic_score_codex":0.002095858,"about_ca_topic_score_gemma":0.0023663277,"teacher_disagreement_score":0.008959019,"about_ca_system_score_codex":0.00062557944,"about_ca_system_score_gemma":0.0005403488,"threshold_uncertainty_score":0.047380447},"labels":[],"label_agreement":null},{"id":"W4243718109","doi":"10.1007/978-1-4939-7131-2_100365","title":"Exemplar-Based Topic Detection","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science","score_opus":0.016549681423278943,"score_gpt":0.24968843038963545,"score_spread":0.2331387489663565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4243718109","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006865682,0.0027928783,0.96999145,0.000269651,0.00069332897,0.00013705093,0.00082315376,0.008172911,0.010253868],"genre_scores_gemma":[0.090289466,0.0037595676,0.8575179,0.00021831965,0.0010064142,0.00020139631,0.0071524037,0.0019307623,0.037923772],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988323,0.00018674525,0.00007213265,0.00036311743,0.00045009062,0.00009560581],"domain_scores_gemma":[0.998281,0.0006248908,0.0000988816,0.00032791667,0.00056303584,0.000104259074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011027614,0.0014293989,0.0012793927,0.003367323,0.0007872142,0.0027527472,0.0020029892,0.0013869342,0.014174278],"category_scores_gemma":[0.0037480246,0.0006008275,0.0011883394,0.0031844035,0.00058197713,0.0026006992,0.0019109589,0.0019287136,0.02264984],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021034088,0.00009762543,0.0005870223,0.00038108596,0.00007727507,0.00017081735,0.0001840314,0.0048089004,0.04868191,0.009657308,0.044499096,0.89064467],"study_design_scores_gemma":[0.000056030785,0.00023413364,0.0038619293,0.00020092397,0.00023978039,0.0023074844,0.00038192174,0.66563356,0.11054481,0.034601144,0.1818065,0.00013180611],"about_ca_topic_score_codex":0.0008799373,"about_ca_topic_score_gemma":0.001235239,"teacher_disagreement_score":0.014174278,"about_ca_system_score_codex":0.00039317395,"about_ca_system_score_gemma":0.00047087556,"threshold_uncertainty_score":0.0474177},"labels":[],"label_agreement":null},{"id":"W4244554860","doi":"10.1007/978-1-4614-6170-8_100407","title":"Topic Identification","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Identification (biology); Computer science; Biology; Botany","score_opus":0.013984898188822318,"score_gpt":0.2539511684350372,"score_spread":0.23996627024621486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4244554860","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0061716754,0.006646382,0.47404116,0.0016232154,0.0030176274,0.0007820381,0.004990704,0.015075796,0.48765135],"genre_scores_gemma":[0.039666284,0.0063352208,0.2224529,0.0006463422,0.0013345729,0.0005747529,0.013213837,0.0038188417,0.71195734],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995141,0.00005831147,0.00003054369,0.00017507102,0.00017263959,0.000049352846],"domain_scores_gemma":[0.99923205,0.00017841608,0.000037991143,0.00017369454,0.00029534686,0.00008257106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00058098946,0.0011177761,0.0008038381,0.0035454973,0.0013534132,0.00319594,0.0010587893,0.0006594298,0.10219714],"category_scores_gemma":[0.0021015392,0.00044520255,0.00090721593,0.0028835821,0.00044729345,0.003417013,0.0023190295,0.0012502046,0.11019554],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000072180635,0.000040345585,0.00030063133,0.00038726753,0.000013014519,0.00007365875,0.00035896577,0.0003410609,0.010313213,0.026060514,0.171927,0.79011214],"study_design_scores_gemma":[0.000012139585,0.00003469289,0.0008963692,0.00019940574,0.000034845252,0.0005757629,0.0003436157,0.0033833573,0.013601061,0.020085264,0.9608046,0.000028851935],"about_ca_topic_score_codex":0.0007041439,"about_ca_topic_score_gemma":0.0010263093,"teacher_disagreement_score":0.10219714,"about_ca_system_score_codex":0.0006678333,"about_ca_system_score_gemma":0.001179629,"threshold_uncertainty_score":0.34188348},"labels":[],"label_agreement":null},{"id":"W4244585827","doi":"10.3410/f.726430947.793521386","title":"Faculty Opinions recommendation of Organizing conceptual knowledge in humans with a gridlike code.","year":2016,"lang":"en","type":"dataset","venue":"Faculty Opinions – Post-Publication Peer Review of the Biomedical Literature","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Baycrest Hospital","funders":"","keywords":"Code (set theory); Computer science; World Wide Web; Knowledge management; Data science; Programming language","score_opus":0.03418532190945529,"score_gpt":0.3600720609278439,"score_spread":0.3258867390183886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4244585827","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0046236473,0.00062158937,0.0045816386,0.0010418611,0.00040171345,0.0003930844,0.9645364,0.010190453,0.013609578],"genre_scores_gemma":[0.008887569,0.00023400543,0.014258037,0.00024768876,0.0000470745,0.0002892636,0.96990126,0.00035141245,0.005783709],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99812645,0.00036302907,0.00025310204,0.0005796084,0.00047467276,0.00020316735],"domain_scores_gemma":[0.99412376,0.0014831884,0.0003674581,0.001867942,0.0016126033,0.00054513256],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00194701,0.0020917035,0.0007496204,0.0054481714,0.0008986219,0.0025981315,0.0022422555,0.0022286484,0.028332518],"category_scores_gemma":[0.015852517,0.000526464,0.0012985065,0.0054187262,0.0005513655,0.002488502,0.002318171,0.001886083,0.03953713],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018468015,0.000109734334,0.0034645514,0.00067640055,0.000037151633,0.000034079905,0.00007049025,0.0006678346,0.00043446178,0.0011727269,0.9681114,0.02503641],"study_design_scores_gemma":[0.00032417796,0.00007815372,0.00641889,0.00043128556,0.00006331146,0.0001213616,0.00029882966,0.012764744,0.0030475103,0.0050940886,0.971306,0.000051659314],"about_ca_topic_score_codex":0.03283469,"about_ca_topic_score_gemma":0.09647092,"teacher_disagreement_score":0.03283469,"about_ca_system_score_codex":0.0015804343,"about_ca_system_score_gemma":0.0040016957,"threshold_uncertainty_score":0.09478176},"labels":[],"label_agreement":null},{"id":"W4245167715","doi":"10.1353/scp.0.0082","title":"Academic Search Engine Optimization ( ASEO ): Optimizing Scholarly Literature for Google Scholar &amp; Co.","year":2010,"lang":"en","type":"article","venue":"Journal of Scholarly Publishing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Class (philosophy); Computer science; Information retrieval; Mathematics; Artificial intelligence","score_opus":0.031066419338067825,"score_gpt":0.3222108468978427,"score_spread":0.2911444275597749,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4245167715","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07315818,0.013628761,0.79024166,0.0049493806,0.0006702591,0.0017809956,0.0037052361,0.011952661,0.09991282],"genre_scores_gemma":[0.26325735,0.00462821,0.708903,0.00038415796,0.0003124931,0.00073609676,0.003095565,0.0016537397,0.01702931],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99621886,0.0013898144,0.0003022829,0.00032493897,0.0015585918,0.00020562203],"domain_scores_gemma":[0.99502677,0.0024523071,0.0004879375,0.0007119206,0.0011026494,0.00021841048],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0037190665,0.0015829434,0.0015691844,0.0068195197,0.00081979064,0.004146679,0.0012839179,0.0010890991,0.005580116],"category_scores_gemma":[0.018727781,0.00047224335,0.0009934822,0.010670331,0.00073838845,0.0044507394,0.0021821568,0.0008434255,0.0033425621],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045327118,0.00035534566,0.004163995,0.0013331639,0.0003166219,0.00012819373,0.0002634959,0.10012932,0.005386842,0.061237313,0.035319135,0.79091316],"study_design_scores_gemma":[0.00020175318,0.00048376797,0.0032455302,0.00029632048,0.00042556316,0.0004077064,0.00050514063,0.77186316,0.016579282,0.11508181,0.09079402,0.00011592717],"about_ca_topic_score_codex":0.0032680286,"about_ca_topic_score_gemma":0.0058395397,"teacher_disagreement_score":0.9958533,"about_ca_system_score_codex":0.001440672,"about_ca_system_score_gemma":0.00405491,"threshold_uncertainty_score":0.01966852},"labels":[{"model":"gemma","categories":["scholarly_communication"],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":["scholarly_communication"],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W4246171214","doi":"10.1007/978-1-4614-6170-8_100791","title":"Social Trends Discovery","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Data science; Computer science","score_opus":0.018764718986644377,"score_gpt":0.28038641787090857,"score_spread":0.2616216988842642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246171214","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00806648,0.015761811,0.5205773,0.008281569,0.002763705,0.00072881207,0.007966294,0.007228609,0.4286254],"genre_scores_gemma":[0.05700126,0.017336069,0.36794376,0.0011247736,0.0022517643,0.0005145335,0.01678623,0.0018201896,0.5352214],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99915373,0.00012128896,0.000040742878,0.00019355412,0.00044500237,0.0000457158],"domain_scores_gemma":[0.9987332,0.00054145243,0.0000667636,0.00024607955,0.00033602674,0.00007651682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011209468,0.0011275751,0.000769481,0.0066447,0.001575032,0.004026754,0.0014176389,0.0009879529,0.035783164],"category_scores_gemma":[0.0039603775,0.0005641725,0.0011103484,0.005905767,0.00063721155,0.0047329986,0.0023875672,0.0014810242,0.026883364],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029843723,0.00004350111,0.00064734893,0.00036810088,0.000050017163,0.00011006637,0.00037205295,0.0010010228,0.0019011751,0.100091875,0.1737062,0.7216788],"study_design_scores_gemma":[0.000008923779,0.000022078459,0.0011484326,0.000252227,0.000052141146,0.0005653888,0.00036581684,0.015521098,0.004978915,0.1811513,0.7958983,0.000035438385],"about_ca_topic_score_codex":0.0017379523,"about_ca_topic_score_gemma":0.0036012,"teacher_disagreement_score":0.035783164,"about_ca_system_score_codex":0.0012795402,"about_ca_system_score_gemma":0.0014221903,"threshold_uncertainty_score":0.11970663},"labels":[],"label_agreement":null},{"id":"W4246558970","doi":"10.36227/techrxiv.12100692","title":"Deep Learning for text in limted data settings","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Computer science; Artificial intelligence; Sequence (biology); Deep learning; Transfer of learning; Sequence learning; Natural language processing; Recurrent neural network; Machine learning; Sentiment analysis; Artificial neural network","score_opus":0.05509319530841789,"score_gpt":0.34010155808589826,"score_spread":0.2850083627774804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246558970","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010233567,0.0027297647,0.9730117,0.003458361,0.00031498462,0.00012590959,0.002297845,0.0028328528,0.004995075],"genre_scores_gemma":[0.4108974,0.0062166196,0.54767805,0.0014419509,0.0009396685,0.0007786052,0.009045507,0.0008122283,0.02218995],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988223,0.00044705864,0.00008738174,0.00029333003,0.00023858252,0.00011128785],"domain_scores_gemma":[0.9972805,0.0016485583,0.0001699138,0.00044433415,0.0003651636,0.000091536436],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024022954,0.0011391237,0.0008336513,0.0012132984,0.0005791528,0.0021047406,0.00188061,0.0023597672,0.0112375235],"category_scores_gemma":[0.010067603,0.0006126419,0.000777169,0.0016427924,0.0009104637,0.0053395596,0.002412104,0.00391023,0.004486683],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035611534,0.00031393697,0.0024259922,0.0009290324,0.00018964655,0.000513524,0.00023171784,0.27146754,0.004538769,0.12387433,0.053488135,0.5416713],"study_design_scores_gemma":[0.000014339498,0.000030067902,0.00030752175,0.00004453251,0.0000080017335,0.000046655434,0.000028728286,0.8908819,0.0011521882,0.09999159,0.007484122,0.000010321687],"about_ca_topic_score_codex":0.005268581,"about_ca_topic_score_gemma":0.0090259295,"teacher_disagreement_score":0.0112375235,"about_ca_system_score_codex":0.0016928852,"about_ca_system_score_gemma":0.0010481193,"threshold_uncertainty_score":0.037593305},"labels":[],"label_agreement":null},{"id":"W4246957940","doi":"10.1002/asi.21222","title":"Individual differences in the interpretation of text: Implications for information science","year":2009,"lang":"en","type":"article","venue":"Journal of the American Society for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Interpretation (philosophy); Computer science; Meaning (existential); Cohesion (chemistry); Natural language processing; Linguistics; Search engine indexing; Information retrieval; Artificial intelligence; Psychology","score_opus":0.013594081527149957,"score_gpt":0.3118840206932907,"score_spread":0.2982899391661407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246957940","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98976326,0.00041561457,0.003920083,0.0004019759,0.000046222773,0.00007508696,0.00011561288,0.000033481,0.0052285516],"genre_scores_gemma":[0.998516,0.00005394943,0.0010351466,0.00007397884,0.000012339405,0.000024593513,0.000036098998,0.000015081042,0.00023267212],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9913993,0.0045346464,0.0005517517,0.0015350193,0.0017865796,0.0001926812],"domain_scores_gemma":[0.87336534,0.10756118,0.007132496,0.0064112935,0.0041523357,0.0013772675],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009909346,0.00027993406,0.0004966157,0.0011549136,0.0005770362,0.0018268775,0.0005096202,0.0005927475,0.0030536547],"category_scores_gemma":[0.091643274,0.0002664075,0.0002634221,0.00078841916,0.00204265,0.0020094116,0.0010308194,0.0007422974,0.00028163826],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026400974,0.0011081693,0.7252228,0.0006816057,0.0011136695,0.0010338781,0.081777714,0.0047660978,0.021800995,0.009041092,0.0030095573,0.14780438],"study_design_scores_gemma":[0.00007452646,0.0006340697,0.94846386,0.000095162395,0.00019649549,0.0008061241,0.013439779,0.0067776726,0.0043339105,0.023195691,0.0018278917,0.00015480239],"about_ca_topic_score_codex":0.0012309431,"about_ca_topic_score_gemma":0.0011503821,"teacher_disagreement_score":0.009909346,"about_ca_system_score_codex":0.00042896305,"about_ca_system_score_gemma":0.0002564535,"threshold_uncertainty_score":0.05240625},"labels":[],"label_agreement":null},{"id":"W4247024223","doi":"10.22360/springsim.2017.cns.009","title":"A Comparative Study On Content-Based Papepr-To-Paper Recommendation Approaches in Scientific Literature","year":2017,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Information retrieval; Set (abstract data type); tf–idf; Word embedding; Word (group theory); Domain (mathematical analysis); Representation (politics); Term (time); Recommender system; Embedding; Data mining; Artificial intelligence; Mathematics","score_opus":0.20708297861746142,"score_gpt":0.37094591894310536,"score_spread":0.16386294032564394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4247024223","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3747458,0.067094475,0.4930841,0.0028834378,0.0009314759,0.0022998583,0.0051979967,0.008652132,0.04511073],"genre_scores_gemma":[0.5793068,0.014263581,0.38777095,0.00044876803,0.00051880674,0.0005804782,0.006471212,0.00030872689,0.010330676],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99045885,0.0028866644,0.000959671,0.0010702875,0.004265548,0.00035895838],"domain_scores_gemma":[0.9715517,0.016385,0.0015827899,0.0030431454,0.006688115,0.0007491994],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0074345563,0.0011709974,0.0012367647,0.018758906,0.0010308385,0.0034342601,0.001761521,0.0018028669,0.003481078],"category_scores_gemma":[0.033802412,0.00039246486,0.0018094546,0.015784554,0.0006239268,0.0052141743,0.0012891939,0.00079675764,0.0027029843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008213617,0.00072630326,0.027194116,0.002388043,0.0010007207,0.0001620407,0.0006490368,0.0057716263,0.008755942,0.004196828,0.005564206,0.94276977],"study_design_scores_gemma":[0.00068735966,0.006026074,0.23641892,0.0016884207,0.0047181994,0.0050225304,0.0066911634,0.5089583,0.07104535,0.02426981,0.13360468,0.0008692604],"about_ca_topic_score_codex":0.0048702983,"about_ca_topic_score_gemma":0.006532871,"teacher_disagreement_score":0.9812411,"about_ca_system_score_codex":0.0013092078,"about_ca_system_score_gemma":0.0020422777,"threshold_uncertainty_score":0.039318204},"labels":[],"label_agreement":null},{"id":"W4247874641","doi":"10.1016/b0-08-044854-2/00963-9","title":"Indexing, Automatic","year":2006,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Search engine indexing; Computer science; Automatic indexing; Information retrieval; Index (typography); Vocabulary; Task (project management); Natural language processing; World Wide Web; Linguistics","score_opus":0.010951203903698793,"score_gpt":0.24686742240760456,"score_spread":0.23591621850390576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4247874641","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0058256383,0.012512296,0.67424095,0.0017892014,0.002821426,0.0007827714,0.014913788,0.056025296,0.2310886],"genre_scores_gemma":[0.04245805,0.008670205,0.5391001,0.0011908266,0.0011994998,0.0006200502,0.04534684,0.0061446293,0.3552697],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99802727,0.00015605094,0.0002034954,0.00053255423,0.00093775324,0.00014295682],"domain_scores_gemma":[0.99864775,0.00021211972,0.0000659546,0.00062256627,0.0004029162,0.000048727223],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011070663,0.0016509591,0.0014946901,0.0058738855,0.002124587,0.006737196,0.0025491829,0.0015220215,0.09969841],"category_scores_gemma":[0.0042000874,0.0010211496,0.0010641012,0.009483021,0.0012259381,0.007985164,0.0032262022,0.001427464,0.1090188],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000061298124,0.000058114714,0.00020246863,0.0006397617,0.000019220004,0.000082333885,0.00014422815,0.00058958464,0.012322977,0.033511937,0.17099836,0.7813698],"study_design_scores_gemma":[0.000039406477,0.00005859896,0.000953696,0.00021787008,0.000058527698,0.0009117544,0.00023342457,0.01198997,0.022219727,0.087892376,0.8753549,0.00006987461],"about_ca_topic_score_codex":0.0038604762,"about_ca_topic_score_gemma":0.0048341216,"teacher_disagreement_score":0.09969841,"about_ca_system_score_codex":0.0010169598,"about_ca_system_score_gemma":0.0023483427,"threshold_uncertainty_score":0.3335244},"labels":[],"label_agreement":null},{"id":"W4247975909","doi":"10.1007/978-1-4614-6170-8_100216","title":"Content-Based Filtering","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science","score_opus":0.046991318907752676,"score_gpt":0.2502640339573551,"score_spread":0.2032727150496024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4247975909","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0043108277,0.003136931,0.9115455,0.00078718015,0.0010500732,0.00045544637,0.0018174384,0.010697506,0.06619905],"genre_scores_gemma":[0.050245438,0.00544841,0.64054954,0.0009852923,0.0011308688,0.00042967987,0.008643811,0.0038867001,0.2886802],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998206,0.00022354773,0.0001192822,0.00045227617,0.00086937577,0.00012949134],"domain_scores_gemma":[0.9969392,0.0009799192,0.000098326855,0.0007573691,0.0011492894,0.000075967255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016290009,0.0016583257,0.0014443218,0.006101522,0.0017803293,0.0038979815,0.001814595,0.0013206871,0.04025087],"category_scores_gemma":[0.004856348,0.0006884234,0.0016417268,0.0054125213,0.0008185359,0.0039577805,0.0018649359,0.0015601001,0.04442873],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013004859,0.00010006285,0.0003740461,0.00037356236,0.000053209107,0.00008017763,0.00019945473,0.0013480379,0.017874544,0.03054135,0.07005096,0.8788745],"study_design_scores_gemma":[0.000051076753,0.00014898834,0.0027716968,0.00041007638,0.00030641467,0.0011222932,0.00032548094,0.082654506,0.09735578,0.10982027,0.7048869,0.00014654985],"about_ca_topic_score_codex":0.0030644988,"about_ca_topic_score_gemma":0.0036864448,"teacher_disagreement_score":0.04025087,"about_ca_system_score_codex":0.0011129004,"about_ca_system_score_gemma":0.0015163594,"threshold_uncertainty_score":0.13465255},"labels":[],"label_agreement":null},{"id":"W4248679402","doi":"10.1016/j.jda.2004.04.010","title":"Editorial","year":2004,"lang":"es","type":"editorial","venue":"Journal of Discrete Algorithms","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science","score_opus":0.007218399082258171,"score_gpt":0.29535023212277245,"score_spread":0.28813183304051426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248679402","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00005761272,0.0023208882,0.0001580938,0.022296654,0.9703461,0.000021748154,0.00006838861,0.000072563154,0.004657938],"genre_scores_gemma":[0.0011478198,0.0027022136,0.00023705607,0.015317649,0.93126905,0.000037968384,0.00009996108,0.00008791803,0.049100358],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.996874,0.0005086057,0.000379044,0.00042249195,0.0014979604,0.0003179184],"domain_scores_gemma":[0.9770697,0.0051969276,0.001753127,0.0011131747,0.0116204405,0.003246739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038544058,0.0025651264,0.0026774716,0.004961308,0.0024023948,0.0060567753,0.0022386091,0.008225981,0.05428324],"category_scores_gemma":[0.026756426,0.00079443585,0.0016078317,0.001689335,0.0014515622,0.002690642,0.0013984705,0.009319812,0.03289975],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024722087,0.000006535257,0.000013830482,0.00006370206,0.00000673149,0.00007607578,0.0000030039096,0.000012318237,0.00003490651,0.00015103858,0.99449855,0.005108585],"study_design_scores_gemma":[0.000042863612,0.000016804534,0.0001815234,0.00015022607,0.000030113542,0.000243784,0.0000191221,0.00010587974,0.00014822053,0.00071803964,0.99833375,0.000009651955],"about_ca_topic_score_codex":0.0008153547,"about_ca_topic_score_gemma":0.0018439868,"teacher_disagreement_score":0.05428324,"about_ca_system_score_codex":0.0019375383,"about_ca_system_score_gemma":0.002219092,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4248953021","doi":"10.4018/9781591406907.ch003.ch000","title":"Dominant Meanings Approach Towards Individualized Web Search for Learning Environments","year":2011,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Meaning (existential); Computer science; Set (abstract data type); The Internet; Information retrieval; Resource (disambiguation); Search engine; World Wide Web; Semantic search; Web search query; Hypermedia; Psychology","score_opus":0.035756497399137215,"score_gpt":0.276887220525089,"score_spread":0.2411307231259518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248953021","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0129159875,0.0028420405,0.9293745,0.0017117129,0.00015915961,0.00031600235,0.00020665883,0.0019108899,0.050563026],"genre_scores_gemma":[0.20387651,0.0026484209,0.7626893,0.0007525081,0.00021059847,0.000640121,0.0007279609,0.0011443485,0.027310234],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9936987,0.0022172136,0.00038210824,0.0006415112,0.0027471366,0.0003132298],"domain_scores_gemma":[0.9957777,0.0024825356,0.00018943957,0.000867143,0.0005619432,0.00012119461],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032323985,0.0011193426,0.0014387469,0.0047491267,0.0021541887,0.006058521,0.0022458984,0.0018461812,0.010292444],"category_scores_gemma":[0.015554669,0.0010216429,0.0016298868,0.0048444187,0.0035887004,0.01991742,0.0049136872,0.0030649742,0.0045120944],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001350252,0.000119073426,0.00050497276,0.0006860429,0.000057274876,0.00021598744,0.004423786,0.006488844,0.006984026,0.6787177,0.011739212,0.28992805],"study_design_scores_gemma":[0.00006461784,0.00008827732,0.000530769,0.00018490638,0.00008392513,0.00068311614,0.0016185526,0.097351685,0.0072149765,0.77410203,0.11797481,0.000102252794],"about_ca_topic_score_codex":0.0026840384,"about_ca_topic_score_gemma":0.003769503,"teacher_disagreement_score":0.010292444,"about_ca_system_score_codex":0.0024728207,"about_ca_system_score_gemma":0.0021950586,"threshold_uncertainty_score":0.034431636},"labels":[],"label_agreement":null},{"id":"W4249665032","doi":"10.1007/978-1-4614-6170-8_110065","title":"Sentiment-Emotion-Intent Analysis","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Sentiment analysis; Psychology; Natural language processing; Computer science","score_opus":0.013139370942909396,"score_gpt":0.2503086469972744,"score_spread":0.237169276054365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4249665032","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0069541135,0.003651905,0.91088265,0.001282635,0.0012368676,0.00031223634,0.0027792405,0.006755593,0.06614473],"genre_scores_gemma":[0.12001092,0.0065472717,0.6724722,0.0009351824,0.0015044298,0.0005618516,0.012658099,0.0025653793,0.18274474],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996145,0.000076726305,0.000026216332,0.00008244888,0.00017184232,0.000028228184],"domain_scores_gemma":[0.9995685,0.00016084492,0.000028116145,0.000048756436,0.0001748808,0.000018836148],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00073575863,0.0011896873,0.0005963021,0.001633502,0.00043819446,0.0018100551,0.00057072076,0.00044908744,0.014321533],"category_scores_gemma":[0.001733033,0.0003207961,0.0010022813,0.0015646277,0.0003153043,0.0016210034,0.0008977725,0.0013337751,0.015684439],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000052331885,0.00006556488,0.000686553,0.00033074772,0.000048675887,0.00009003246,0.00022141004,0.0014631504,0.01689317,0.029229878,0.093472846,0.8574458],"study_design_scores_gemma":[0.000031744574,0.00014753682,0.009387899,0.0004418869,0.00019568896,0.0010547535,0.0006988516,0.23118734,0.05280144,0.19057947,0.5133466,0.00012678662],"about_ca_topic_score_codex":0.000660259,"about_ca_topic_score_gemma":0.0011960821,"teacher_disagreement_score":0.014321533,"about_ca_system_score_codex":0.00048359434,"about_ca_system_score_gemma":0.00040016606,"threshold_uncertainty_score":0.047910333},"labels":[],"label_agreement":null},{"id":"W4250002556","doi":"10.1007/978-1-4939-7131-2_100048","title":"Automatic Document Topic Identification","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Identification (biology); Computer science; Information retrieval; Natural language processing; World Wide Web; Biology","score_opus":0.013628153498928994,"score_gpt":0.27614683076420066,"score_spread":0.26251867726527167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4250002556","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015699117,0.0140989395,0.8742151,0.001056742,0.0022418727,0.00027966048,0.0038484028,0.019409543,0.06915064],"genre_scores_gemma":[0.1204863,0.009904166,0.6290028,0.0004130009,0.0016794846,0.0003599818,0.019917298,0.0032310744,0.21500587],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994129,0.000084004954,0.000040864754,0.00018064126,0.00022230636,0.00005933885],"domain_scores_gemma":[0.99891365,0.0003473915,0.00006182104,0.0001935508,0.00042893607,0.000054675784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006574857,0.001067889,0.00080818223,0.003978011,0.0008391104,0.002403491,0.00083293027,0.00072418334,0.018168934],"category_scores_gemma":[0.0018949657,0.00044615543,0.0008250259,0.0028887049,0.00035544846,0.002236963,0.0013070307,0.0011355544,0.024230119],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000102371494,0.00005109891,0.00038423316,0.000403134,0.000029321982,0.000093159295,0.00014524911,0.0008228535,0.034625813,0.005837599,0.07982178,0.8776833],"study_design_scores_gemma":[0.00004144091,0.00014905198,0.005101288,0.00036302852,0.0002093701,0.0022669751,0.0004619368,0.08362454,0.15162893,0.032139532,0.72389793,0.00011603265],"about_ca_topic_score_codex":0.00058122736,"about_ca_topic_score_gemma":0.0008824129,"teacher_disagreement_score":0.018168934,"about_ca_system_score_codex":0.00047899006,"about_ca_system_score_gemma":0.0006562262,"threshold_uncertainty_score":0.06078112},"labels":[],"label_agreement":null},{"id":"W4250951234","doi":"10.32920/ryerson.14645355","title":"Microblog summarization based on sentiment and aspect analysis","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Wilfrid Laurier University; Toronto Metropolitan University; University of Waterloo","funders":"","keywords":"Automatic summarization; Microblogging; Sentiment analysis; Social media; Computer science; Information retrieval; Baseline (sea); Cluster analysis; Multi-document summarization; Annotation; Data science; Natural language processing; Artificial intelligence; World Wide Web","score_opus":0.009684912507483769,"score_gpt":0.263804563741241,"score_spread":0.25411965123375724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4250951234","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23870215,0.0022083344,0.73147196,0.0008962441,0.0005006314,0.0013315363,0.0070774886,0.009661342,0.008150413],"genre_scores_gemma":[0.4250402,0.0014330064,0.54698664,0.00015836112,0.0009456891,0.00070387445,0.015612593,0.00077276997,0.008346947],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99946004,0.00010482223,0.00007159553,0.0001094962,0.00019141313,0.00006259992],"domain_scores_gemma":[0.9978855,0.000507356,0.00030271927,0.00015940992,0.0010581348,0.00008685203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006930597,0.0011361658,0.0007982749,0.0044519263,0.0006023171,0.0013150509,0.00044450132,0.00037874424,0.001614424],"category_scores_gemma":[0.0031018977,0.00025651904,0.0006801919,0.0027339673,0.00018613116,0.0013495969,0.000694063,0.0005549684,0.0014884702],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000607295,0.00020175875,0.0071421512,0.00079994247,0.00019738352,0.00041616257,0.0013060757,0.007150688,0.14122717,0.0023016327,0.017940827,0.820709],"study_design_scores_gemma":[0.00015935613,0.0012597425,0.050518114,0.00019798496,0.00085694797,0.0010839642,0.0036612467,0.6009088,0.24294461,0.016769845,0.081438705,0.00020065582],"about_ca_topic_score_codex":0.0016736678,"about_ca_topic_score_gemma":0.0030687512,"teacher_disagreement_score":0.0044519263,"about_ca_system_score_codex":0.00033973585,"about_ca_system_score_gemma":0.00045703942,"threshold_uncertainty_score":0.005400777},"labels":[],"label_agreement":null},{"id":"W4251063661","doi":"10.1007/978-1-4939-7131-2_101113","title":"Social Indexing","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Search engine indexing; Computer science; Information retrieval","score_opus":0.029386760224444655,"score_gpt":0.2976400481211125,"score_spread":0.26825328789666786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251063661","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002958257,0.012324273,0.13081937,0.0041261874,0.0051060016,0.00063020043,0.009452088,0.011205952,0.82337767],"genre_scores_gemma":[0.026928479,0.015608677,0.08545565,0.0013786085,0.0049393084,0.00051504327,0.023153728,0.0031416162,0.8388789],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987909,0.00015585357,0.00008738284,0.00018656309,0.0006977962,0.00008149312],"domain_scores_gemma":[0.99868625,0.0002967937,0.00006693783,0.0004253275,0.00042377523,0.000100864825],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008241962,0.0012142018,0.000962031,0.007296417,0.002009685,0.0052594524,0.0014563431,0.0009725764,0.17478655],"category_scores_gemma":[0.0038466216,0.00042301306,0.00088542595,0.009046621,0.00087624136,0.0069498383,0.0034019595,0.0011346537,0.15146922],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024680106,0.000043326258,0.00009517475,0.0004387183,0.000015240075,0.000030942032,0.000116431744,0.00028568358,0.0024900874,0.055095002,0.3490339,0.5923309],"study_design_scores_gemma":[0.000006172639,0.000015818754,0.00023295108,0.00011566787,0.000012889828,0.0001790126,0.00009630518,0.0014471135,0.0020005421,0.04281057,0.9530656,0.000017363629],"about_ca_topic_score_codex":0.0013777509,"about_ca_topic_score_gemma":0.002677962,"teacher_disagreement_score":0.17478655,"about_ca_system_score_codex":0.0012208383,"about_ca_system_score_gemma":0.0015940472,"threshold_uncertainty_score":0.5847193},"labels":[],"label_agreement":null},{"id":"W4251440222","doi":"10.1007/978-1-4939-7131-2_100201","title":"Content-Based Filtering","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science","score_opus":0.06038741613714324,"score_gpt":0.26616976127405756,"score_spread":0.20578234513691432,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251440222","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004861393,0.0029880235,0.90821797,0.000908213,0.0012918594,0.00052311795,0.0022373882,0.012402412,0.06656963],"genre_scores_gemma":[0.05654744,0.0052156164,0.6397143,0.0011491965,0.0013998656,0.00050293334,0.010333852,0.004487797,0.28064904],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99797624,0.00026069747,0.00013925655,0.0005201804,0.00095697667,0.00014662277],"domain_scores_gemma":[0.996182,0.0011933509,0.00012164384,0.00097554404,0.0014319218,0.00009553482],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018047427,0.0017472202,0.0014872334,0.006485433,0.001930916,0.004255482,0.0018083957,0.0014136032,0.04238569],"category_scores_gemma":[0.0058179707,0.00070803665,0.0017852249,0.0055921255,0.0008084537,0.0041313856,0.0020037547,0.0017243925,0.048521224],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016028169,0.00012544928,0.00045609433,0.00044744246,0.00006459127,0.00009220193,0.00022997444,0.001498518,0.02041837,0.03343147,0.080413826,0.86266184],"study_design_scores_gemma":[0.000055923832,0.00016035509,0.0028375168,0.00042968738,0.0003369342,0.0010690523,0.00033189292,0.086914174,0.09890864,0.105053164,0.7037538,0.00014879978],"about_ca_topic_score_codex":0.0029660452,"about_ca_topic_score_gemma":0.0035337564,"teacher_disagreement_score":0.04238569,"about_ca_system_score_codex":0.0011398987,"about_ca_system_score_gemma":0.0016860234,"threshold_uncertainty_score":0.14179426},"labels":[],"label_agreement":null},{"id":"W4253452607","doi":"10.1007/978-1-4939-7131-2_101061","title":"Sentiment-Emotion-Intent Analysis","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Sentiment analysis; Psychology; Computer science; Cognitive psychology; Natural language processing","score_opus":0.016776002897686302,"score_gpt":0.26446226045971055,"score_spread":0.24768625756202425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4253452607","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007720555,0.003915388,0.9067571,0.0015622893,0.0017140409,0.00033017976,0.0034465718,0.007829405,0.0667245],"genre_scores_gemma":[0.12787926,0.0068678977,0.65882427,0.0011017651,0.0020653566,0.0006016144,0.016314875,0.0030889423,0.18325609],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958044,0.00008326023,0.000028950566,0.00009388091,0.00018215214,0.00003120919],"domain_scores_gemma":[0.9994703,0.00018852313,0.000034927325,0.000063868305,0.00021679333,0.000025615742],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081521994,0.0012075584,0.0006043366,0.0016414885,0.00045079362,0.0019277142,0.0005859947,0.0004819636,0.016000172],"category_scores_gemma":[0.0021030318,0.00031645864,0.0010795302,0.001570398,0.0003148177,0.0017652015,0.0009858077,0.0014897039,0.018030198],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000647669,0.000078728124,0.0007490596,0.00038502135,0.00005423024,0.00009790625,0.00024166994,0.0014772261,0.018666884,0.031635676,0.115609296,0.83093953],"study_design_scores_gemma":[0.000033042797,0.00015168556,0.008754011,0.0004531207,0.000197547,0.0009522628,0.0006445311,0.22220537,0.049778443,0.18364792,0.5330584,0.00012362316],"about_ca_topic_score_codex":0.00060514495,"about_ca_topic_score_gemma":0.0010545257,"teacher_disagreement_score":0.016000172,"about_ca_system_score_codex":0.0004896517,"about_ca_system_score_gemma":0.0004185263,"threshold_uncertainty_score":0.053525925},"labels":[],"label_agreement":null},{"id":"W4254172015","doi":"10.1007/978-1-4939-7131-2_101361","title":"Topic Identification","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Identification (biology); Computer science; Biology; Botany","score_opus":0.018762735002870085,"score_gpt":0.2712372913040963,"score_spread":0.2524745563012262,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254172015","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006188181,0.0065154536,0.48312062,0.0018473802,0.003629339,0.0008207258,0.0058817635,0.017514572,0.4744819],"genre_scores_gemma":[0.042388357,0.0063662017,0.23818502,0.00076711306,0.0016874251,0.0006693052,0.015940418,0.0045332615,0.68946296],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99943644,0.00006909832,0.000036651534,0.00020611721,0.00019571594,0.000055830234],"domain_scores_gemma":[0.99907255,0.00021248801,0.00004646468,0.00021718167,0.0003518432,0.00009943008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006686287,0.0011627426,0.00083304895,0.0037795785,0.0014029385,0.003468198,0.0010943229,0.0007053561,0.10903662],"category_scores_gemma":[0.002497754,0.0004581967,0.0009690865,0.0030109687,0.0004448135,0.0035677636,0.0024352218,0.0013460367,0.123847716],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008116013,0.000045938417,0.00031673696,0.00043184237,0.00001486591,0.000076461474,0.00035817415,0.0003617291,0.010749069,0.02651285,0.19000277,0.7710485],"study_design_scores_gemma":[0.000012893797,0.00003591526,0.00086304056,0.00020192929,0.00003643627,0.000543215,0.00031764962,0.0035411364,0.013488837,0.01983081,0.9610985,0.00002964124],"about_ca_topic_score_codex":0.0006358324,"about_ca_topic_score_gemma":0.0009046566,"teacher_disagreement_score":0.10903662,"about_ca_system_score_codex":0.0006860734,"about_ca_system_score_gemma":0.0012489425,"threshold_uncertainty_score":0.36476386},"labels":[],"label_agreement":null},{"id":"W4254928945","doi":"10.32920/ryerson.14649648.v1","title":"Fuzzy Thesauri Recommendation System For Web 2.0 Social networks","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Information retrieval; World Wide Web; The Internet; Scalability; Set (abstract data type); Node (physics); Social network (sociolinguistics); Order (exchange); Domain (mathematical analysis); Database; Social media","score_opus":0.026919574988326124,"score_gpt":0.30151589840289234,"score_spread":0.2745963234145662,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254928945","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24606709,0.0035536531,0.6984055,0.0013102315,0.0004804528,0.0014825914,0.0033890428,0.020503664,0.024807787],"genre_scores_gemma":[0.72590476,0.0012022608,0.24851313,0.00024090582,0.000121485245,0.00041412367,0.0022910805,0.00011011984,0.021201978],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948007,0.000076448974,0.000058233018,0.00012816745,0.00020741823,0.000049658564],"domain_scores_gemma":[0.9995382,0.000085774474,0.00003533854,0.0000530206,0.0002546419,0.000033025233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071749487,0.00062438817,0.00098458,0.0018768954,0.0009912295,0.0010421263,0.0011365943,0.0011140615,0.005861525],"category_scores_gemma":[0.001658992,0.00026077055,0.00084305473,0.0010971663,0.00019312662,0.0010306421,0.00040275176,0.0005550573,0.0024740677],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023568005,0.0011008905,0.008562291,0.00067976065,0.00038207797,0.0010685807,0.00040790113,0.065736935,0.040917303,0.0060545946,0.031941094,0.84079176],"study_design_scores_gemma":[0.00012870177,0.0002455945,0.003705674,0.00003086843,0.00014602121,0.00025377065,0.00013189665,0.9733208,0.012252106,0.0018203134,0.007899892,0.00006442048],"about_ca_topic_score_codex":0.046441548,"about_ca_topic_score_gemma":0.055436008,"teacher_disagreement_score":0.046441548,"about_ca_system_score_codex":0.0012370893,"about_ca_system_score_gemma":0.0009285171,"threshold_uncertainty_score":0.092342496},"labels":[],"label_agreement":null},{"id":"W4255165398","doi":"10.4018/978-1-59140-441-5.ch008","title":"KEA","year":2004,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; Information retrieval; Metadata; Feature (linguistics); Artificial intelligence; License; Natural language processing; World Wide Web; Linguistics","score_opus":0.014728942334229727,"score_gpt":0.2612281464799091,"score_spread":0.24649920414567938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4255165398","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003851584,0.0109422775,0.22814302,0.0016647509,0.0012630213,0.0003405833,0.017812198,0.05447026,0.6815124],"genre_scores_gemma":[0.017816953,0.007363598,0.14934339,0.00088469795,0.0003416782,0.00023921575,0.022794208,0.011049459,0.7901669],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99957234,0.00004527823,0.000035284043,0.00011460468,0.00019906467,0.000033337583],"domain_scores_gemma":[0.99897194,0.0003174786,0.00004559111,0.00026466796,0.00031600392,0.00008430014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004770194,0.0012109218,0.0006368831,0.0035319,0.0008218603,0.004018272,0.0013208367,0.001123618,0.19551268],"category_scores_gemma":[0.0021700913,0.0005691048,0.0006532796,0.0033584565,0.00048359873,0.0062928055,0.0015403574,0.0014059626,0.23239093],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009068071,0.00004375,0.00027986633,0.0007653131,0.000017497136,0.00012033974,0.0003720821,0.00050413475,0.006673384,0.060432095,0.3464848,0.58421606],"study_design_scores_gemma":[0.000004255417,0.000008106397,0.00021130209,0.00006583202,0.000005091426,0.00026328437,0.000040420815,0.0005522855,0.001515178,0.006155426,0.9911696,0.000009309269],"about_ca_topic_score_codex":0.00091943715,"about_ca_topic_score_gemma":0.0019678823,"teacher_disagreement_score":0.19551268,"about_ca_system_score_codex":0.00093710655,"about_ca_system_score_gemma":0.00084979244,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4256499950","doi":"10.1007/978-1-4614-6170-8_100872","title":"Social Indexing","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Search engine indexing; Computer science; Information retrieval","score_opus":0.022375741178368,"score_gpt":0.28043612194115036,"score_spread":0.25806038076278237,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4256499950","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030560794,0.012682546,0.13861594,0.003572803,0.004091125,0.00059943675,0.007437291,0.009727082,0.82021767],"genre_scores_gemma":[0.02776947,0.0153407,0.08932459,0.0011814905,0.00393473,0.000476206,0.01796447,0.002593051,0.84141517],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989459,0.00013474382,0.00007215077,0.00015823406,0.0006174399,0.000071530456],"domain_scores_gemma":[0.9989379,0.00024698637,0.000055254197,0.00033500756,0.0003477732,0.00007712511],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007340239,0.0011223487,0.00091541006,0.0066023106,0.0019260288,0.0046000145,0.0013893998,0.00088500086,0.14949593],"category_scores_gemma":[0.003187887,0.00040185955,0.00078741525,0.008255658,0.0008903171,0.006366384,0.0030982736,0.0010438865,0.121077985],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002354793,0.000039906226,0.00009177386,0.0003889086,0.000013589289,0.000031404143,0.000120783196,0.00030044562,0.0023612797,0.060358666,0.31541267,0.62085706],"study_design_scores_gemma":[0.000006354241,0.0000162692,0.00024270355,0.000115815375,0.00001272236,0.00020049824,0.000107684566,0.0015848199,0.0022115698,0.04847632,0.94700736,0.0000179367],"about_ca_topic_score_codex":0.0015327082,"about_ca_topic_score_gemma":0.0029835722,"teacher_disagreement_score":0.14949593,"about_ca_system_score_codex":0.0012060566,"about_ca_system_score_gemma":0.0014597478,"threshold_uncertainty_score":0.5001137},"labels":[],"label_agreement":null},{"id":"W4288751120","doi":"10.5539/ijel.v12n5p59","title":"Teaching Complex Sentences in ESL Reading: Structural Analysis","year":2022,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Jilin Office of Philosophy and Social Science","keywords":"Conceptualization; Reading (process); Test (biology); Second language; Cognition; Linguistics; Psychology; Empirical research; Mathematics education; Computer science; Artificial intelligence; Mathematics","score_opus":0.019093918799176143,"score_gpt":0.32839580407688956,"score_spread":0.3093018852777134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288751120","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99730116,0.000017585788,0.001036494,0.00004001326,0.0000011897783,0.000020261068,0.000026313082,0.000010114293,0.0015469025],"genre_scores_gemma":[0.9977591,0.000025063175,0.0017192845,0.000007403045,0.0000013924962,0.00001788327,0.000051591644,0.0000038448748,0.0004142535],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9994692,0.0002151378,0.00003068321,0.00011981876,0.00011864755,0.00004637686],"domain_scores_gemma":[0.995331,0.0030720087,0.0006920475,0.00025179217,0.00045976884,0.00019329229],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007041922,0.00031475857,0.00017084899,0.00084939296,0.00038206275,0.0008506504,0.00031778286,0.00020048356,0.0024830583],"category_scores_gemma":[0.0053473227,0.00015730991,0.00019279616,0.00074115413,0.00078208937,0.00093086,0.00067102903,0.00040151837,0.00023823054],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005614075,0.0028387166,0.50590473,0.00047822483,0.0000847681,0.0010819907,0.07491588,0.0033301606,0.0733819,0.010939641,0.00095511833,0.32552737],"study_design_scores_gemma":[0.000075222495,0.0012812184,0.9080068,0.00007566433,0.00012589412,0.00028274627,0.026868144,0.01832647,0.030677916,0.010421219,0.0038084623,0.000050292394],"about_ca_topic_score_codex":0.004994231,"about_ca_topic_score_gemma":0.011460691,"teacher_disagreement_score":0.004994231,"about_ca_system_score_codex":0.00090069853,"about_ca_system_score_gemma":0.0014313639,"threshold_uncertainty_score":0.009930372},"labels":[],"label_agreement":null},{"id":"W4292200231","doi":"10.7287/peerj-cs.1066v0.2/reviews/2","title":"Peer Review #2 of \"Causal graph extraction from news: a comparative study of time-series causality learning techniques (v0.2)\"","year":2022,"lang":"en","type":"peer-review","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Causality (physics); Series (stratigraphy); Graph; Computer science; Time series; Psychology; Data science; Theoretical computer science; Machine learning; Physics; Geology","score_opus":0.06287932553139104,"score_gpt":0.4000561116054687,"score_spread":0.3371767860740777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4292200231","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016345425,0.036552966,0.056613818,0.17380063,0.42117456,0.008576705,0.028533634,0.013645761,0.24475646],"genre_scores_gemma":[0.07074017,0.043918703,0.046976138,0.02457097,0.09431779,0.0051747654,0.055430286,0.010236507,0.64863473],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9828523,0.0032909934,0.001366315,0.0014542724,0.010064247,0.0009719626],"domain_scores_gemma":[0.7912475,0.022792783,0.0066230274,0.014425601,0.15589733,0.009013744],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.014250947,0.0010285685,0.0020386986,0.007194408,0.0047031534,0.008713019,0.004377766,0.0032140086,0.24424893],"category_scores_gemma":[0.13475184,0.0008214371,0.001619026,0.0063303495,0.00148091,0.0054569268,0.0057167523,0.0024079564,0.13206607],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000082540326,0.00003233821,0.00075221184,0.0014528636,0.00004588229,0.00009872765,0.00008907969,0.000211072,0.0006724863,0.0013038758,0.8699966,0.12526232],"study_design_scores_gemma":[0.00004996917,0.00005035537,0.002101125,0.0006356168,0.000047299483,0.00013128977,0.00018727944,0.0012450296,0.0010758472,0.0022339043,0.99220467,0.00003770436],"about_ca_topic_score_codex":0.00426189,"about_ca_topic_score_gemma":0.013007494,"teacher_disagreement_score":0.98574907,"about_ca_system_score_codex":0.0020052944,"about_ca_system_score_gemma":0.010777951,"threshold_uncertainty_score":0.8170941},"labels":[],"label_agreement":null},{"id":"W4293863311","doi":"10.1109/siu55565.2022.9864851","title":"Automatic Keyword Extraction From Dialogue Text","year":2022,"lang":"en","type":"article","venue":"2022 30th Signal Processing and Communications Applications Conference (SIU)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Stantec (Canada)","funders":"","keywords":"Computer science; Keyword extraction; Dialog box; Recall; Word (group theory); Natural language processing; Precision and recall; Information retrieval; Process (computing); Artificial intelligence; Customer service; Service (business); World Wide Web; Linguistics","score_opus":0.029836366513141047,"score_gpt":0.3031218038835432,"score_spread":0.2732854373704022,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293863311","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16846547,0.005114078,0.76865697,0.0005564686,0.000639866,0.0013536207,0.017553259,0.026354803,0.0113054],"genre_scores_gemma":[0.30965522,0.001464702,0.6540188,0.00017225115,0.00025345734,0.0009554672,0.022287834,0.0011559793,0.010036329],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99783236,0.00048867724,0.00038990218,0.0004925539,0.00059691374,0.00019953723],"domain_scores_gemma":[0.99498874,0.0016184743,0.00033616976,0.00035599712,0.0025646302,0.00013593242],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011188638,0.0014900047,0.0012796182,0.005000246,0.00070550444,0.0014037618,0.0006564389,0.00080146774,0.0052738967],"category_scores_gemma":[0.0061641084,0.00032686646,0.0007601778,0.002734729,0.00027005115,0.0018761007,0.0008948096,0.00045975117,0.008960374],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008693155,0.00011426076,0.0049721613,0.0026343458,0.000089192574,0.0010187575,0.0014653642,0.0014712026,0.36141756,0.0022446022,0.01738705,0.60631615],"study_design_scores_gemma":[0.00023336145,0.00096638594,0.029367376,0.0004994653,0.00037059106,0.006241049,0.0037347039,0.10995796,0.66102886,0.009053927,0.17817388,0.00037243907],"about_ca_topic_score_codex":0.0012572716,"about_ca_topic_score_gemma":0.0012240998,"teacher_disagreement_score":0.0052738967,"about_ca_system_score_codex":0.0004123262,"about_ca_system_score_gemma":0.0010841624,"threshold_uncertainty_score":0.017642975},"labels":[],"label_agreement":null},{"id":"W4297993723","doi":"10.1007/978-1-4614-7163-9_352-1","title":"Automatic Document Topic Identification Using Social Knowledge Network","year":2017,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Identification (biology); Computer science; Social knowledge; Data science; Information retrieval; Sociology; Social science; Biology","score_opus":0.04112207874473995,"score_gpt":0.33929912789038236,"score_spread":0.2981770491456424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4297993723","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032341786,0.01241196,0.9245307,0.0007065158,0.00097472774,0.00024922172,0.002437304,0.0074198763,0.018927876],"genre_scores_gemma":[0.23386209,0.0076312865,0.7013189,0.00020884824,0.001178635,0.00031066584,0.00967251,0.0010665313,0.0447505],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992668,0.00013090394,0.000057006473,0.0002023705,0.00027829848,0.00006461942],"domain_scores_gemma":[0.99880266,0.0005823872,0.000094404146,0.00014631532,0.00032369472,0.000050435716],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081083836,0.00083500234,0.00077876734,0.005109175,0.0007964563,0.002261735,0.00076101895,0.000779837,0.0045738923],"category_scores_gemma":[0.0025284379,0.00034978322,0.0008411695,0.0040311087,0.00034442294,0.0031296562,0.0010879962,0.000911504,0.0058335746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013756032,0.00008012024,0.0013051678,0.00041471337,0.00006668738,0.00014789784,0.0002756807,0.0028664484,0.033158157,0.006403577,0.024917873,0.93022615],"study_design_scores_gemma":[0.000063447194,0.00020079965,0.012829198,0.00037712068,0.0004459126,0.002629179,0.0009690785,0.57344383,0.12899366,0.062219694,0.21764241,0.0001857118],"about_ca_topic_score_codex":0.0012945555,"about_ca_topic_score_gemma":0.0018475457,"teacher_disagreement_score":0.005109175,"about_ca_system_score_codex":0.0006069697,"about_ca_system_score_gemma":0.00059508823,"threshold_uncertainty_score":0.015301168},"labels":[],"label_agreement":null},{"id":"W4298195797","doi":"","title":"1 Exact versus Estimated Pruning of Subject Hierarchies","year":2016,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Pruning; Subject (documents); Computer science; Artificial intelligence; Horticulture; World Wide Web; Biology","score_opus":0.03281499216325376,"score_gpt":0.3134140566580464,"score_spread":0.28059906449479266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4298195797","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.304464,0.016397776,0.5469073,0.0045159776,0.001293621,0.00065544137,0.037374094,0.034795083,0.053596586],"genre_scores_gemma":[0.5049209,0.001597758,0.44632092,0.0007254821,0.00046261484,0.00018237837,0.028357113,0.0020944409,0.015338433],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995167,0.0016602651,0.0004007823,0.0011301727,0.0011418074,0.0004999229],"domain_scores_gemma":[0.9878149,0.007509425,0.00045019868,0.002673363,0.0012958855,0.00025624796],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003581002,0.0009434498,0.0011444172,0.0033568367,0.0008750674,0.003918613,0.0021653355,0.001975572,0.0139695825],"category_scores_gemma":[0.02374903,0.00052191404,0.00091849244,0.0031362146,0.0007169122,0.0038337384,0.0017288465,0.001147494,0.0041598985],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030268552,0.0003526294,0.013258937,0.0015102656,0.0004299892,0.0003129133,0.00060988724,0.074453235,0.012867006,0.02831932,0.10845891,0.7564001],"study_design_scores_gemma":[0.0005300187,0.00048761998,0.016936472,0.00054237613,0.00053688587,0.0013039093,0.00064104324,0.8303201,0.018630603,0.07081979,0.05914289,0.000108242406],"about_ca_topic_score_codex":0.009722156,"about_ca_topic_score_gemma":0.029695613,"teacher_disagreement_score":0.0139695825,"about_ca_system_score_codex":0.0012011428,"about_ca_system_score_gemma":0.0033665295,"threshold_uncertainty_score":0.046732843},"labels":[],"label_agreement":null},{"id":"W4298371492","doi":"","title":"Output Keywords in Context in an HTML File with Python","year":2012,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Python (programming language); Computer science; Programming language; Operating system; Information retrieval; Database","score_opus":0.22058783022283282,"score_gpt":0.5482749587316242,"score_spread":0.32768712850879145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4298371492","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003410532,0.00012229908,0.18295027,0.0007915954,0.0004991528,0.00031916754,0.08128892,0.7046844,0.025933703],"genre_scores_gemma":[0.07994647,0.00067481835,0.3502009,0.0024568574,0.00040787808,0.0019408772,0.15096688,0.32297674,0.090428606],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99949193,0.000046573237,0.00004968203,0.0001534384,0.00018229344,0.00007607271],"domain_scores_gemma":[0.9986424,0.0005073207,0.00009370251,0.0002514089,0.00039452943,0.000110683286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006796307,0.0019714357,0.0009138336,0.0010046887,0.00055314624,0.0015485599,0.0018394275,0.0010479066,0.19536234],"category_scores_gemma":[0.005264719,0.00088535046,0.0013401104,0.001027226,0.00041266927,0.0028899973,0.0029691758,0.0018373829,0.11953462],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008932705,0.00018851197,0.0021431472,0.0014728159,0.00008595419,0.0005773375,0.00037282254,0.0017838341,0.008483965,0.006653171,0.82718134,0.15016387],"study_design_scores_gemma":[0.0005466792,0.00018805669,0.005232001,0.00052520557,0.000108427266,0.0007419604,0.0003534416,0.036306128,0.06465589,0.050959192,0.8401351,0.00024794787],"about_ca_topic_score_codex":0.0016483323,"about_ca_topic_score_gemma":0.0018199534,"teacher_disagreement_score":0.19536234,"about_ca_system_score_codex":0.00062282383,"about_ca_system_score_gemma":0.0011020735,"threshold_uncertainty_score":0.6535522},"labels":[],"label_agreement":null},{"id":"W4299291826","doi":"","title":"Application of Trend Detection Methods in Monitoring Physiological Signals","year":2004,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada","funders":"","keywords":"Computer science","score_opus":0.2818971117291039,"score_gpt":0.615143491350574,"score_spread":0.3332463796214701,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4299291826","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051898472,0.0028949168,0.9389381,0.00024742153,0.00016318884,0.00027943615,0.00023023237,0.001109772,0.004238523],"genre_scores_gemma":[0.29289988,0.00388477,0.70034087,0.00008440989,0.00014105561,0.00024207802,0.00042417677,0.00015502481,0.0018276853],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968779,0.0010286588,0.0002325555,0.00035817403,0.0014124757,0.0000902036],"domain_scores_gemma":[0.9907708,0.006206435,0.0005727531,0.00037427468,0.0019942638,0.000081459715],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046440894,0.00085208594,0.000658033,0.0072341715,0.0004733823,0.0012983067,0.000790105,0.00090903474,0.0013637681],"category_scores_gemma":[0.01546385,0.00032070527,0.000765965,0.0036110105,0.00035204674,0.0015683412,0.00052344875,0.00055735017,0.000572597],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023559325,0.00013143025,0.01044126,0.0007072858,0.00025167663,0.00016479199,0.00053421257,0.013283723,0.01975279,0.003027431,0.001010054,0.9504598],"study_design_scores_gemma":[0.00020381704,0.0025776627,0.0769312,0.0009216771,0.0010870299,0.0035250885,0.0018817163,0.7128092,0.121234745,0.02198445,0.05634971,0.0004937562],"about_ca_topic_score_codex":0.0014906718,"about_ca_topic_score_gemma":0.0014392212,"teacher_disagreement_score":0.0072341715,"about_ca_system_score_codex":0.00041041928,"about_ca_system_score_gemma":0.0005046391,"threshold_uncertainty_score":0.02456063},"labels":[],"label_agreement":null},{"id":"W4299335947","doi":"10.1007/978-3-031-01880-0_3","title":"Mining Text Conversations","year":2011,"lang":"en","type":"book-chapter","venue":"Synthesis lectures on data management","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Linguistics; Philosophy","score_opus":0.060711542777487804,"score_gpt":0.2734736238599594,"score_spread":0.21276208108247158,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4299335947","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10855678,0.011801819,0.6825737,0.007726374,0.0022389917,0.0012558886,0.07636155,0.024415907,0.085068956],"genre_scores_gemma":[0.36014363,0.0052816723,0.38664845,0.00079733884,0.0017084646,0.001232661,0.16482066,0.002167135,0.07720003],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.997063,0.0007827566,0.00024273088,0.00085770205,0.0008469111,0.00020701649],"domain_scores_gemma":[0.995748,0.002223713,0.00036150767,0.000608586,0.0008220085,0.00023633418],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013977888,0.0013411427,0.0008849389,0.00908979,0.0018657562,0.004453789,0.0014578021,0.0011935123,0.017118758],"category_scores_gemma":[0.010551365,0.00060604105,0.0012798383,0.0059401486,0.0006387776,0.0069671962,0.0026175217,0.0017602175,0.021757407],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002453155,0.00025204784,0.008763437,0.0006819373,0.0001549539,0.00042272426,0.0015368538,0.0021877224,0.011064535,0.018906828,0.106993556,0.8487901],"study_design_scores_gemma":[0.00007612995,0.00028905168,0.024029497,0.0012250774,0.00041167715,0.0022181405,0.009530745,0.1902312,0.050107535,0.19104557,0.5306367,0.00019868182],"about_ca_topic_score_codex":0.0016593933,"about_ca_topic_score_gemma":0.0024737325,"teacher_disagreement_score":0.017118758,"about_ca_system_score_codex":0.00080688053,"about_ca_system_score_gemma":0.0013238739,"threshold_uncertainty_score":0.057267904},"labels":[],"label_agreement":null},{"id":"W4300501538","doi":"10.1007/978-1-4614-6170-8_352","title":"Automatic Document Topic Identification Using Social Knowledge Network","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Identification (biology); Computer science; Social knowledge; Data science; Social network analysis; Information retrieval; World Wide Web; Social media; Sociology; Social science; Biology","score_opus":0.02634131939976245,"score_gpt":0.30789046175082113,"score_spread":0.28154914235105866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4300501538","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03334928,0.011100315,0.92740464,0.0006214094,0.000772721,0.0002499347,0.0021782434,0.0067780707,0.017545385],"genre_scores_gemma":[0.22959706,0.0069352584,0.71149415,0.00017891222,0.00096194335,0.00030054006,0.008299864,0.0009240968,0.04130817],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993044,0.00012680814,0.00005456437,0.00018537816,0.00026693844,0.00006183437],"domain_scores_gemma":[0.9988632,0.0005695159,0.0000857999,0.00013079753,0.0003052597,0.000045342003],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007873857,0.0007921393,0.00076160626,0.0050093285,0.00078781496,0.0020949596,0.0007569538,0.00073270896,0.004183094],"category_scores_gemma":[0.002324291,0.00034908912,0.0007986819,0.0038882757,0.00033739817,0.002950641,0.0009805622,0.00082385505,0.0049294033],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013461345,0.00007720828,0.0013310561,0.00039178162,0.000062450796,0.00014531896,0.0002802339,0.0027910476,0.032723207,0.005940984,0.02143651,0.9346856],"study_design_scores_gemma":[0.000066620414,0.00020300229,0.013710271,0.00036540828,0.0004507988,0.0027595717,0.0010368517,0.5859797,0.13226452,0.059317388,0.20365866,0.00018725934],"about_ca_topic_score_codex":0.0013740232,"about_ca_topic_score_gemma":0.0019861225,"teacher_disagreement_score":0.0050093285,"about_ca_system_score_codex":0.0005898471,"about_ca_system_score_gemma":0.00054817006,"threshold_uncertainty_score":0.0139938},"labels":[],"label_agreement":null},{"id":"W4307128390","doi":"10.1007/s11042-022-14043-z","title":"Systematic review of content analysis algorithms based on deep neural networks","year":2022,"lang":"en","type":"article","venue":"Multimedia Tools and Applications","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning; Deep learning; Artificial neural network; Support vector machine; Subject (documents); The Internet; Algorithm; Data mining; World Wide Web","score_opus":0.0293709047354469,"score_gpt":0.2849881205764562,"score_spread":0.2556172158410093,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307128390","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027381622,0.97725964,0.015172908,0.0010969893,0.00041220116,0.0006287355,0.0014039794,0.0001438089,0.0011435676],"genre_scores_gemma":[0.038644042,0.9060372,0.048251588,0.0018865577,0.000498804,0.0014769654,0.002119843,0.0001331811,0.00095172215],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.9944459,0.0021817875,0.0014625612,0.0005625114,0.0012717813,0.000075423784],"domain_scores_gemma":[0.9597787,0.03233106,0.002952245,0.0010016169,0.0037150313,0.00022135334],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009454068,0.0014388222,0.0033857906,0.007349172,0.0004089842,0.0020853123,0.0018700564,0.000979822,0.004044699],"category_scores_gemma":[0.059662726,0.0005788628,0.004440099,0.0045301085,0.0007859277,0.002523675,0.0013226023,0.0011511901,0.00068986334],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005176707,0.00009086742,0.001183168,0.2459158,0.006140777,0.000040764655,0.000113119124,0.0014866478,0.0007920369,0.0013694564,0.0078373905,0.7345124],"study_design_scores_gemma":[0.0016459931,0.0020644746,0.017736629,0.6214876,0.087737545,0.0008305345,0.0005499478,0.012324853,0.006821696,0.017700566,0.23076186,0.00033834463],"about_ca_topic_score_codex":0.00503689,"about_ca_topic_score_gemma":0.013822919,"teacher_disagreement_score":0.009454068,"about_ca_system_score_codex":0.0015871782,"about_ca_system_score_gemma":0.008230899,"threshold_uncertainty_score":0.04999852},"labels":[],"label_agreement":null},{"id":"W4307260180","doi":"10.5539/ells.v12n4p29","title":"A Study on Lexical Chunks in Different Moves of Abstracts of Native and Chinese Applied Linguistic Journals","year":2022,"lang":"en","type":"article","venue":"English Language and Literature Studies","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Norm (philosophy); Perspective (graphical); Linguistics; Computer science; Natural language processing; Significant difference; Artificial intelligence; Psychology; Mathematics; Epistemology; Statistics; Philosophy","score_opus":0.011140211545487234,"score_gpt":0.3197886199857754,"score_spread":0.3086484084402882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307260180","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9906906,0.0007548673,0.0019649542,0.0002482561,0.0000435324,0.00013786642,0.0005628541,0.00003886939,0.00555828],"genre_scores_gemma":[0.99285567,0.00040404347,0.0041557243,0.00004453913,0.000043651762,0.00017144372,0.00066592835,0.00002596468,0.0016329612],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99471235,0.0015855341,0.0013688881,0.0005841792,0.0015407769,0.00020825867],"domain_scores_gemma":[0.9466725,0.03211394,0.009662294,0.0015230654,0.008769018,0.0012591645],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.003967743,0.00025952962,0.0003773317,0.011632242,0.0020021296,0.003537629,0.00043895663,0.0004249887,0.0021482296],"category_scores_gemma":[0.04451581,0.0001796607,0.00033214228,0.011610351,0.0012980656,0.0033037427,0.0016064744,0.00037069214,0.00038571117],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013486512,0.00020437695,0.28862613,0.0036218355,0.00025877095,0.0018859546,0.41185716,0.0002770786,0.04620573,0.0071681375,0.003103808,0.23544228],"study_design_scores_gemma":[0.000048330767,0.00047781982,0.77518743,0.00044889885,0.00025779518,0.0014986302,0.18411976,0.0015643114,0.007031723,0.00273773,0.026457302,0.00017021778],"about_ca_topic_score_codex":0.00547327,"about_ca_topic_score_gemma":0.008750842,"teacher_disagreement_score":0.98836774,"about_ca_system_score_codex":0.0016399808,"about_ca_system_score_gemma":0.0020319992,"threshold_uncertainty_score":0.020983696},"labels":[],"label_agreement":null},{"id":"W4307381099","doi":"10.1115/1.4056076","title":"A Hybrid Semantic Networks Construction Framework for Engineering Design","year":2022,"lang":"en","type":"article","venue":"Journal of Mechanical Design","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Bombardier (Canada); Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Word2vec; Information retrieval; Natural language processing; Artificial intelligence; Key (lock); Phrase; Thesaurus; Parsing","score_opus":0.024840125948536396,"score_gpt":0.2625807066213537,"score_spread":0.23774058067281728,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307381099","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017646305,0.00012811969,0.9940951,0.00014532327,0.000018177807,0.000095699324,0.00064751453,0.0016493098,0.0014560425],"genre_scores_gemma":[0.06556241,0.00041760414,0.92711985,0.00010451878,0.000029528597,0.00047444197,0.0040484457,0.00029372398,0.001949488],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988129,0.000373086,0.00012344363,0.00029661684,0.0003366508,0.00005728707],"domain_scores_gemma":[0.999102,0.0003821971,0.00010594688,0.00018213302,0.00018858114,0.000039215938],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013120658,0.0014826137,0.0005335492,0.0044897273,0.00066842994,0.0014998706,0.0013308186,0.0008089434,0.004223079],"category_scores_gemma":[0.0033268805,0.000602301,0.0029890395,0.0024952283,0.00080829614,0.0029635811,0.0018182055,0.0012982436,0.0015427726],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000092196846,0.00014026642,0.0020060216,0.0009470092,0.00020082369,0.00056989194,0.0008306663,0.25221017,0.0072345166,0.314547,0.011208222,0.41001323],"study_design_scores_gemma":[0.000016354403,0.000041330397,0.0004281431,0.0001676394,0.00006352555,0.0001553786,0.00014936773,0.80088466,0.0034874007,0.15028825,0.044289816,0.000028108196],"about_ca_topic_score_codex":0.007014671,"about_ca_topic_score_gemma":0.010775485,"teacher_disagreement_score":0.007014671,"about_ca_system_score_codex":0.0015280945,"about_ca_system_score_gemma":0.0016508641,"threshold_uncertainty_score":0.014127612},"labels":[],"label_agreement":null},{"id":"W4307873956","doi":"10.32604/cmc.2023.026607","title":"Identification and Visualization of Spatial and Temporal Trends in Textile Industry","year":2022,"lang":"en","type":"article","venue":"Computers, materials & continua/Computers, materials & continua (Print)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Moncton","funders":"Natural Sciences and Engineering Research Council of Canada; New Brunswick Innovation Foundation","keywords":"Computer science; Data science; Visualization; Phrase; Field (mathematics); Identification (biology); Key (lock); Subject (documents); Textile; Information retrieval; Artificial intelligence; World Wide Web; Geography","score_opus":0.011372053583374715,"score_gpt":0.2688125094646664,"score_spread":0.25744045588129166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307873956","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8291529,0.008109954,0.02610788,0.004590424,0.00043059033,0.00020201775,0.092111215,0.005957166,0.0333379],"genre_scores_gemma":[0.9232411,0.0037441324,0.043702837,0.00012893403,0.00020341278,0.00018887642,0.022682196,0.0002901724,0.0058183945],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99972683,0.000054225482,0.000041339135,0.00005445121,0.00007901116,0.000044142053],"domain_scores_gemma":[0.9976609,0.0007584249,0.0006937972,0.0001409027,0.0006180911,0.00012790733],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070202386,0.000302747,0.00020346907,0.013283335,0.0003214868,0.0015595633,0.00020069642,0.00036504742,0.0034095647],"category_scores_gemma":[0.0031173755,0.00013859331,0.00031726097,0.012050884,0.00016886,0.0013657334,0.0006557704,0.00039523575,0.00087748625],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009051155,0.00019789675,0.36138114,0.0022304638,0.0002671401,0.0019759869,0.0144714825,0.009284291,0.03262967,0.016214363,0.08362398,0.47681847],"study_design_scores_gemma":[0.00004839809,0.0002263734,0.75857,0.000667446,0.00016951012,0.0010384655,0.011931202,0.065633856,0.008679468,0.008383281,0.14455435,0.000097567885],"about_ca_topic_score_codex":0.010394179,"about_ca_topic_score_gemma":0.01675969,"teacher_disagreement_score":0.013283335,"about_ca_system_score_codex":0.0004019532,"about_ca_system_score_gemma":0.0006437135,"threshold_uncertainty_score":0.020667374},"labels":[],"label_agreement":null},{"id":"W4310007486","doi":"10.1109/acii55700.2022.9953882","title":"Choose or Fuse: Enriching Data Views with Multi-label Emotion Dynamics","year":2022,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Fuse (electrical); Dynamics (music); Computer science; Artificial intelligence; Human–computer interaction; Psychology; Electrical engineering; Engineering","score_opus":0.10012331492029614,"score_gpt":0.3502092798871958,"score_spread":0.25008596496689967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4310007486","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07556729,0.0006694053,0.91160625,0.0009087317,0.00017453155,0.00022698476,0.003432189,0.0056996904,0.0017149445],"genre_scores_gemma":[0.38693964,0.00042976407,0.6044412,0.0003565509,0.00014736332,0.00043893937,0.0043311995,0.001197784,0.0017173833],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99813,0.00076931185,0.000092512244,0.000599291,0.0002801979,0.0001286321],"domain_scores_gemma":[0.990893,0.005606086,0.00052115216,0.0019508841,0.0007096875,0.00031917577],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051328354,0.0012502981,0.0007941081,0.001957641,0.0005742829,0.00278501,0.001121099,0.0014457557,0.0025851377],"category_scores_gemma":[0.018760076,0.0005267923,0.0010657752,0.0012013859,0.0008498572,0.0062774024,0.0038392942,0.0022821135,0.0010399573],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030807052,0.0006690361,0.022207366,0.0011004839,0.00041228125,0.00077536236,0.011695765,0.041771285,0.11362827,0.020699127,0.01868822,0.7652721],"study_design_scores_gemma":[0.00008775593,0.00037172783,0.013672359,0.00020661295,0.00014629106,0.00033363473,0.0028695385,0.8091391,0.027633263,0.11680962,0.02852898,0.00020113582],"about_ca_topic_score_codex":0.0011263217,"about_ca_topic_score_gemma":0.0029757894,"teacher_disagreement_score":0.0051328354,"about_ca_system_score_codex":0.0005171137,"about_ca_system_score_gemma":0.000494873,"threshold_uncertainty_score":0.027145386},"labels":[],"label_agreement":null},{"id":"W4312527591","doi":"10.1007/978-3-031-19745-1_12","title":"Measuring the Big Five Factors from Handwriting Using Ensemble Learning Model AvgMlSC","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Handwriting; Artificial intelligence","score_opus":0.05032433373247846,"score_gpt":0.26457773012523367,"score_spread":0.2142533963927552,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312527591","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.74539477,0.0015746969,0.24175145,0.0002777963,0.00022612436,0.00013255098,0.0024219428,0.0019388638,0.006281811],"genre_scores_gemma":[0.9485562,0.00028335507,0.04510999,0.000041484735,0.000050341885,0.00008030553,0.0024510496,0.00007439552,0.003352783],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998987,0.00018357733,0.00008791562,0.00027524628,0.000348668,0.00011761449],"domain_scores_gemma":[0.9962566,0.0015151359,0.0003048838,0.00035821152,0.0013807341,0.00018440468],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014402124,0.0010777515,0.000697771,0.002985651,0.00057073985,0.0015462134,0.0005035743,0.0006869186,0.0022757591],"category_scores_gemma":[0.005037203,0.00019383668,0.00090022234,0.0022795561,0.0002793723,0.0013192168,0.0007564415,0.0006684813,0.0012203994],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008257008,0.00037042948,0.1773362,0.00022691298,0.00064520683,0.00024461336,0.0004152039,0.04806749,0.01891144,0.0007286604,0.006712881,0.7455153],"study_design_scores_gemma":[0.000016804623,0.00052175246,0.19050853,0.00007717722,0.00033729142,0.00028554662,0.00046762248,0.7871368,0.015786896,0.0020613316,0.0026946065,0.00010570707],"about_ca_topic_score_codex":0.0062294137,"about_ca_topic_score_gemma":0.00871015,"teacher_disagreement_score":0.0062294137,"about_ca_system_score_codex":0.00040958132,"about_ca_system_score_gemma":0.00057367346,"threshold_uncertainty_score":0.012386262},"labels":[],"label_agreement":null},{"id":"W4312876730","doi":"10.2139/ssrn.4241291","title":"Language Models for Automated Market Research: A New Way to Generate Perceptual Maps","year":2022,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Perception; Computer science; Cognitive science; Artificial intelligence; Natural language processing; Data science; Psychology; Neuroscience","score_opus":0.03884540882691364,"score_gpt":0.34335700007021086,"score_spread":0.30451159124329724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312876730","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006297624,0.00015094195,0.98253274,0.00043329896,0.00009696735,0.000072741226,0.0009871563,0.003737526,0.0056909868],"genre_scores_gemma":[0.37545156,0.00038789614,0.6140887,0.00031355768,0.00013092562,0.00039856613,0.0019216513,0.0013739082,0.005933215],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993273,0.00026719263,0.000042029907,0.00016250563,0.00014841727,0.000052514395],"domain_scores_gemma":[0.9967886,0.001925176,0.00013840609,0.0005464834,0.00047889166,0.00012239328],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010229786,0.00087201933,0.00052414974,0.0017707859,0.0007028352,0.0026872926,0.0012095263,0.0011716556,0.013594837],"category_scores_gemma":[0.0095482,0.00057223724,0.001441358,0.001333372,0.0007539004,0.004654783,0.0018284086,0.0017567739,0.0035432864],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046961746,0.00030815895,0.0022774194,0.0004902361,0.00020410177,0.00033799704,0.0010149175,0.12998655,0.009892438,0.26590398,0.03457779,0.55453676],"study_design_scores_gemma":[0.000031773845,0.00004454439,0.00034296254,0.00005051751,0.000034090244,0.0000600957,0.00015488872,0.7194641,0.0029167486,0.26628932,0.010584178,0.000026785876],"about_ca_topic_score_codex":0.003423244,"about_ca_topic_score_gemma":0.004590561,"teacher_disagreement_score":0.013594837,"about_ca_system_score_codex":0.000865369,"about_ca_system_score_gemma":0.0011177049,"threshold_uncertainty_score":0.045479238},"labels":[],"label_agreement":null},{"id":"W4312991542","doi":"10.1115/detc2022-90688","title":"Knowledge Extraction Method to Support Domain Integrated Design Methodology","year":2022,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; McGill University","funders":"","keywords":"Computer science; Domain (mathematical analysis); Domain knowledge; Focus (optics); Artificial intelligence; Machine learning; Data mining","score_opus":0.14133252813912373,"score_gpt":0.44001695740530955,"score_spread":0.2986844292661858,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312991542","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014763417,0.00027629835,0.97150934,0.000381505,0.0000679432,0.00071560405,0.0019364364,0.0065128724,0.003836667],"genre_scores_gemma":[0.093362026,0.00022631764,0.8982417,0.0001661118,0.000037387756,0.0007740461,0.0049129887,0.00015049831,0.0021289133],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9975235,0.00065174856,0.0004183939,0.00061586435,0.00069513044,0.00009537087],"domain_scores_gemma":[0.99486524,0.00241363,0.0004110577,0.00057943724,0.0016353219,0.000095283875],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00227243,0.0010363859,0.0007494543,0.005875975,0.0007449545,0.0020601186,0.0012071726,0.0010534694,0.0046685156],"category_scores_gemma":[0.0077745165,0.00032981831,0.0014281812,0.002739071,0.00037907148,0.0021993166,0.001088781,0.00097463693,0.0023057284],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023009405,0.00048444993,0.0031458305,0.0008970744,0.00012588123,0.00041441846,0.00048189168,0.012492932,0.026085394,0.008490301,0.008457244,0.9386944],"study_design_scores_gemma":[0.00016306034,0.00036514952,0.0044804146,0.00056468585,0.00033567686,0.0007206927,0.0007160444,0.76813364,0.09412882,0.037403703,0.0928859,0.00010222579],"about_ca_topic_score_codex":0.0016106057,"about_ca_topic_score_gemma":0.0018626498,"teacher_disagreement_score":0.005875975,"about_ca_system_score_codex":0.0009562306,"about_ca_system_score_gemma":0.0019093863,"threshold_uncertainty_score":0.015617728},"labels":[],"label_agreement":null},{"id":"W4316664928","doi":"10.21203/rs.3.rs-2428155/v1","title":"EmoAtlas: An emotional profiling tool merging psychological lexicons, artificial intelligence and network science","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Profiling (computer programming); Emotional intelligence; Psychological science; Artificial intelligence; Computer science; Psychology; Data science; Cognitive science; Social psychology","score_opus":0.23669094474095342,"score_gpt":0.49690815325457166,"score_spread":0.26021720851361824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4316664928","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025319044,0.0005674007,0.7106654,0.0005728525,0.00053193816,0.0005861921,0.031924214,0.21032771,0.019505182],"genre_scores_gemma":[0.25743204,0.0010484474,0.6163397,0.0010665873,0.00036637482,0.0020794668,0.06734417,0.019703308,0.03461989],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947304,0.000144195,0.00007460206,0.00010140317,0.00017284072,0.000033941848],"domain_scores_gemma":[0.998142,0.0011449751,0.00015830481,0.00019385814,0.00026124704,0.000099564044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007554423,0.0011596096,0.00055268727,0.003271306,0.0005739182,0.0025774683,0.0008336326,0.0006435816,0.018568423],"category_scores_gemma":[0.00547211,0.00047910673,0.0006777963,0.0017862006,0.00024046557,0.0030297674,0.0018017344,0.0010271725,0.0110867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010320657,0.00038681805,0.006691799,0.001841622,0.00026004654,0.0005843062,0.0013726766,0.0047946996,0.034473594,0.027332956,0.24078988,0.6804396],"study_design_scores_gemma":[0.0003019423,0.0003740071,0.016752535,0.00051535654,0.00046981423,0.0015982757,0.0011374044,0.28930178,0.07639659,0.11309466,0.4997512,0.00030640457],"about_ca_topic_score_codex":0.0009342457,"about_ca_topic_score_gemma":0.0019132083,"teacher_disagreement_score":0.018568423,"about_ca_system_score_codex":0.00036830793,"about_ca_system_score_gemma":0.00059866905,"threshold_uncertainty_score":0.062117577},"labels":[],"label_agreement":null},{"id":"W4320558549","doi":"10.48550/arxiv.2302.04983","title":"CREDENCE: Counterfactual Explanations for Document Ranking","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University; York University; University of Waterloo","funders":"","keywords":"Credence; Counterfactual thinking; Ranking (information retrieval); Information retrieval; Computer science; Attribution; Machine learning; Psychology; Social psychology","score_opus":0.13831414081683752,"score_gpt":0.24837989918099043,"score_spread":0.11006575836415292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320558549","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0077803503,0.0003708972,0.9823199,0.0022014263,0.0001597604,0.00010037588,0.0007966839,0.0022783391,0.003992277],"genre_scores_gemma":[0.45590794,0.0007964464,0.53296757,0.0010381708,0.0007256979,0.0007213715,0.0021107486,0.00086453534,0.004867556],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.991839,0.005114394,0.0003339185,0.0008128894,0.0016049687,0.00029492105],"domain_scores_gemma":[0.9289482,0.06249204,0.0023217392,0.004397428,0.0011948872,0.0006456078],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011212886,0.0013963884,0.0010241376,0.0039784363,0.0015870362,0.004711864,0.003094997,0.0033347209,0.018299667],"category_scores_gemma":[0.07876221,0.0008031069,0.002430128,0.0024774347,0.0035330816,0.00737447,0.0044628684,0.004353998,0.0016330524],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004578817,0.00016553473,0.0030780768,0.00039962388,0.00018977365,0.00075410254,0.0009846159,0.1131754,0.0014985326,0.7856458,0.015154882,0.07849586],"study_design_scores_gemma":[0.000053954598,0.000026859223,0.0002490732,0.00005060673,0.000029459447,0.0001284726,0.00005929439,0.33568946,0.0010887966,0.655733,0.0068541113,0.00003686451],"about_ca_topic_score_codex":0.002412322,"about_ca_topic_score_gemma":0.0028353648,"teacher_disagreement_score":0.018299667,"about_ca_system_score_codex":0.0019751724,"about_ca_system_score_gemma":0.0017886184,"threshold_uncertainty_score":0.0612185},"labels":[],"label_agreement":null},{"id":"W4321488394","doi":"10.1145/3539597.3573033","title":"DisKeyword: Tweet Corpora Exploration for Keyword Selection","year":2023,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Social media; Selection (genetic algorithm); Information retrieval; Task (project management); License; GRASP; Keyword search; Artificial intelligence; World Wide Web; Natural language processing","score_opus":0.04859705573421192,"score_gpt":0.32179649944986105,"score_spread":0.27319944371564914,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4321488394","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035430674,0.003311289,0.5734131,0.0016997631,0.00096803595,0.0008203447,0.030843766,0.34199533,0.0115177855],"genre_scores_gemma":[0.09027897,0.0010688617,0.8498138,0.00053177343,0.00026550915,0.00094820425,0.029038364,0.015758324,0.0122962305],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990049,0.00021020861,0.000094275776,0.00030822086,0.00028916728,0.00009322561],"domain_scores_gemma":[0.99695146,0.0017428526,0.00016285162,0.0005117885,0.0004574721,0.00017364536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014814691,0.0020045158,0.0013912083,0.0047632246,0.0012477564,0.0027730141,0.0016620173,0.0011456202,0.021287587],"category_scores_gemma":[0.010827576,0.0007941422,0.0011636385,0.0041490328,0.00053886086,0.004434672,0.002912487,0.0012685572,0.018519994],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012284417,0.00027057566,0.004854908,0.001989566,0.0003841594,0.00048465902,0.001413228,0.005031469,0.036387514,0.007676304,0.21884269,0.7214365],"study_design_scores_gemma":[0.00082519255,0.00043511786,0.005536894,0.00032651663,0.00024197892,0.0012832929,0.0018310411,0.5754772,0.072556935,0.043645766,0.2975868,0.00025329227],"about_ca_topic_score_codex":0.0030034375,"about_ca_topic_score_gemma":0.008902439,"teacher_disagreement_score":0.021287587,"about_ca_system_score_codex":0.00061298296,"about_ca_system_score_gemma":0.001537076,"threshold_uncertainty_score":0.07121408},"labels":[],"label_agreement":null},{"id":"W4322630579","doi":"10.21203/rs.3.rs-2497596/v1","title":"Systematic Review using a Spiral approach with Machine Learning","year":2023,"lang":"en","type":"review","venue":"Research Square","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Spiral (railway); Computer science; Artificial intelligence; Engineering; Mechanical engineering","score_opus":0.21997874520230526,"score_gpt":0.4883235181318533,"score_spread":0.26834477292954806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4322630579","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008881044,0.36052167,0.45168087,0.0069624037,0.001781584,0.1390952,0.014480976,0.0025707835,0.014025532],"genre_scores_gemma":[0.038197417,0.08169723,0.81556517,0.0011795775,0.00033943364,0.058524128,0.0017435852,0.00015644779,0.0025970363],"study_design_codex":"systematic_review","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9082281,0.05507418,0.02288879,0.0035228569,0.009745151,0.0005410148],"domain_scores_gemma":[0.68779624,0.25997007,0.024873804,0.007691794,0.018306699,0.0013613527],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06516188,0.0027617917,0.009030784,0.034151264,0.0019288145,0.00601871,0.0028431679,0.0019310373,0.020753015],"category_scores_gemma":[0.2693986,0.0021716757,0.009949595,0.025960468,0.0019514322,0.005738596,0.004640192,0.0024735124,0.0024831542],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059363886,0.000103919774,0.0014832652,0.6211729,0.010715216,0.00023581118,0.0015882773,0.003599393,0.0006640924,0.01609835,0.008711695,0.33503333],"study_design_scores_gemma":[0.0030574538,0.002136905,0.0043577584,0.6243931,0.10494827,0.00082332874,0.0028060973,0.027539877,0.0022307688,0.097276814,0.13010736,0.00032220947],"about_ca_topic_score_codex":0.0043027126,"about_ca_topic_score_gemma":0.015468504,"teacher_disagreement_score":0.9348381,"about_ca_system_score_codex":0.007008506,"about_ca_system_score_gemma":0.03404019,"threshold_uncertainty_score":0.34461308},"labels":[],"label_agreement":null},{"id":"W4322751637","doi":"10.1108/ejm-07-2020-0542","title":"A comparative study of the predictive power of component-based approaches to structural equation modeling","year":2022,"lang":"en","type":"article","venue":"European Journal of Marketing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; McGill University","funders":"","keywords":"Structural equation modeling; Component (thermodynamics); Computer science; Predictive power; Predictive modelling; Econometrics; Path analysis (statistics); Partial least squares regression; Machine learning; Mathematics","score_opus":0.10331513278271896,"score_gpt":0.2758739453762855,"score_spread":0.17255881259356654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4322751637","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5764509,0.008882685,0.38507584,0.0043621273,0.00048052182,0.0012503439,0.0007071898,0.001983815,0.0208065],"genre_scores_gemma":[0.8947133,0.0026106013,0.100361064,0.00023730914,0.00017181464,0.00051386526,0.00064642495,0.00018842095,0.0005571758],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9458847,0.045601103,0.00093539525,0.00195126,0.005261087,0.00036639953],"domain_scores_gemma":[0.63378125,0.33710834,0.0051385257,0.008970607,0.013894489,0.0011066825],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07857579,0.0023269742,0.001222015,0.008296901,0.0009423129,0.005024462,0.0022476579,0.0012896119,0.0026688043],"category_scores_gemma":[0.22387317,0.0008192221,0.0023664865,0.009625831,0.0020413606,0.005878329,0.0024053073,0.0026696979,0.0005580463],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015034887,0.00086518895,0.17827073,0.0015499112,0.0027800146,0.00015525095,0.0061681257,0.11020129,0.0006018198,0.06871952,0.0048651565,0.6243195],"study_design_scores_gemma":[0.0002336014,0.0014309493,0.051906317,0.0012257618,0.0009922272,0.00017783072,0.0024810412,0.87625617,0.00095864665,0.06043109,0.0037000685,0.00020631026],"about_ca_topic_score_codex":0.0075937747,"about_ca_topic_score_gemma":0.0057469653,"teacher_disagreement_score":0.07857579,"about_ca_system_score_codex":0.0020392353,"about_ca_system_score_gemma":0.0034119245,"threshold_uncertainty_score":0.41555345},"labels":[],"label_agreement":null},{"id":"W4322764463","doi":"10.1038/s41597-023-02015-3","title":"The Three Terms Task - an open benchmark to compare human and artificial semantic representations","year":2023,"lang":"en","type":"article","venue":"Scientific Data","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Institut Universitaire de Gériatrie de Montréal","funders":"","keywords":"Computer science; Semantic memory; Natural language processing; Artificial intelligence; Semantic similarity; Task (project management); Benchmark (surveying); Word (group theory); Similarity (geometry); Associative property; Noun; Representation (politics); Cognition; Linguistics; Psychology; Mathematics","score_opus":0.14608141974920205,"score_gpt":0.4256313126982627,"score_spread":0.27954989294906063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4322764463","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.50011706,0.005151339,0.070388794,0.0031528617,0.0018169576,0.0032098165,0.33369517,0.006973332,0.0754946],"genre_scores_gemma":[0.38206822,0.000840344,0.087405026,0.0013145242,0.00042530734,0.005145132,0.5089397,0.0013636689,0.012498082],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99551857,0.0015837745,0.0007601236,0.00090956927,0.0010482455,0.00017965032],"domain_scores_gemma":[0.98647004,0.006318623,0.0012365766,0.0033092555,0.00186306,0.00080243155],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034268654,0.0015816364,0.0009776242,0.0031136316,0.0016348641,0.002119375,0.0022247508,0.0026981079,0.008412775],"category_scores_gemma":[0.024098983,0.00028676412,0.0015892129,0.0028478885,0.001257949,0.004622595,0.0035434782,0.0020872273,0.007981929],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004965624,0.0027097582,0.072603606,0.007860576,0.0010509333,0.0014070534,0.005018494,0.011631368,0.027510576,0.025012452,0.46404728,0.37618217],"study_design_scores_gemma":[0.0014762176,0.002583086,0.16224457,0.0011318498,0.0004218051,0.0050915377,0.006430395,0.07181837,0.02435546,0.1175997,0.6062579,0.00058912835],"about_ca_topic_score_codex":0.0033046359,"about_ca_topic_score_gemma":0.0057385005,"teacher_disagreement_score":0.008412775,"about_ca_system_score_codex":0.0010142202,"about_ca_system_score_gemma":0.0012960562,"threshold_uncertainty_score":0.028143525},"labels":[],"label_agreement":null},{"id":"W4323315086","doi":"10.1002/jcpy.1346","title":"Style, content, and the success of ideas","year":2023,"lang":"en","type":"article","venue":"Journal of Consumer Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Style (visual arts); Function (biology); Perspective (graphical); Context (archaeology); Psychology; Content (measure theory); Class (philosophy); Value (mathematics); Variance (accounting); Natural (archaeology); Writing style; Social psychology; Cognitive psychology; Linguistics; Computer science; Artificial intelligence","score_opus":0.04070871232479893,"score_gpt":0.3619696436976745,"score_spread":0.32126093137287554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323315086","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94952273,0.0011610668,0.0068539046,0.0013732715,0.00016404182,0.00014553798,0.00026653544,0.00018189767,0.040331047],"genre_scores_gemma":[0.99371815,0.00043699789,0.0033149498,0.0001317839,0.0002042708,0.000058293004,0.00015916758,0.00009304967,0.0018833631],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9898699,0.0044279937,0.00080314715,0.0004879168,0.0041551013,0.0002559742],"domain_scores_gemma":[0.702185,0.23291908,0.032304686,0.006508756,0.021343475,0.0047390023],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013878096,0.0005813526,0.0005375781,0.007240307,0.001342165,0.011116031,0.00070431305,0.0010018164,0.0066655553],"category_scores_gemma":[0.16644667,0.00032092456,0.000614486,0.006067527,0.002680648,0.0057278997,0.0024931189,0.0016251318,0.0014749083],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021551605,0.0013254748,0.48358762,0.002129195,0.0010483499,0.0013617947,0.041658025,0.003765196,0.016226044,0.03436579,0.0077428734,0.40463445],"study_design_scores_gemma":[0.0004720823,0.0026436665,0.7793398,0.0015249194,0.0010314382,0.0014261763,0.02438787,0.024744371,0.017656047,0.10946883,0.03696975,0.00033494073],"about_ca_topic_score_codex":0.00041854984,"about_ca_topic_score_gemma":0.00040697888,"teacher_disagreement_score":0.013878096,"about_ca_system_score_codex":0.0014875982,"about_ca_system_score_gemma":0.0008473045,"threshold_uncertainty_score":0.07339525},"labels":[],"label_agreement":null},{"id":"W4360764647","doi":"10.1109/icmla55696.2022.00027","title":"Towards Emotion Cause Generation in Natural Language Processing using Deep Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Task (project management); Computer science; Generative grammar; Artificial intelligence; Natural language processing; Emotion recognition; Task analysis; Emotion classification; Deep learning; Cognitive psychology; Speech recognition; Psychology; Engineering","score_opus":0.02281344021177964,"score_gpt":0.312059707008795,"score_spread":0.28924626679701537,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4360764647","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01614974,0.00068718527,0.9787466,0.0008212973,0.000086511245,0.00009225635,0.00019715507,0.0017956302,0.0014236392],"genre_scores_gemma":[0.4358482,0.00092826085,0.55614454,0.00070716685,0.00023522892,0.00025824163,0.0012777612,0.000301869,0.004298828],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991725,0.0003608902,0.00005261704,0.0002042538,0.00014083575,0.000068912945],"domain_scores_gemma":[0.99780244,0.0016151454,0.00013585377,0.00017231797,0.00021964728,0.00005449618],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017216213,0.00089498225,0.00047282607,0.0010987258,0.00045291492,0.001325446,0.0008881066,0.0009420369,0.0025168876],"category_scores_gemma":[0.0042490996,0.00035240132,0.0011647225,0.00070983276,0.00080201135,0.0024470089,0.0013593906,0.0023630194,0.0009117055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003341114,0.00029820186,0.002837854,0.00070571154,0.00017649839,0.00050802046,0.001115166,0.091200665,0.042317584,0.050177984,0.013972204,0.796356],"study_design_scores_gemma":[0.000024602146,0.000054945955,0.0005465249,0.000030005136,0.000031095904,0.00006624525,0.00011047525,0.8978716,0.008209084,0.089475214,0.0035645205,0.000015626103],"about_ca_topic_score_codex":0.0013497389,"about_ca_topic_score_gemma":0.0027377817,"teacher_disagreement_score":0.0025168876,"about_ca_system_score_codex":0.0007007422,"about_ca_system_score_gemma":0.0007229263,"threshold_uncertainty_score":0.0091049075},"labels":[],"label_agreement":null},{"id":"W4360789531","doi":"10.7202/1097460ar","title":"Approche computationnelle de l’analyse conceptuelle","year":2023,"lang":"fr","type":"article","venue":"Philosophiques","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Humanities; Philosophy; Physics","score_opus":0.28500287423019793,"score_gpt":0.39873394825107233,"score_spread":0.1137310740208744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4360789531","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001984631,0.00030765825,0.994232,0.00039142813,0.000064649765,0.00005915562,0.000078045,0.0010489095,0.0018336457],"genre_scores_gemma":[0.067447715,0.0005605331,0.92600983,0.00021914729,0.00012005235,0.00026792934,0.00031679813,0.00034155304,0.004716456],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9927326,0.0026309106,0.00037357153,0.0015412214,0.0024291973,0.00029247117],"domain_scores_gemma":[0.98490834,0.010493866,0.00038974642,0.0017048542,0.0023419426,0.00016120505],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005306416,0.0014728941,0.0013708632,0.0035976516,0.0015487338,0.007095283,0.0025299462,0.0021078265,0.015178596],"category_scores_gemma":[0.023260886,0.0011864743,0.0025978298,0.0027482584,0.0025115816,0.007647267,0.0026260691,0.004194041,0.004927149],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052694883,0.00015336725,0.0011525943,0.0011312863,0.0003018617,0.00021171178,0.0014556173,0.045945715,0.019989472,0.31377643,0.008623749,0.6067312],"study_design_scores_gemma":[0.000097171,0.0000802593,0.0007901073,0.0002803706,0.00011520023,0.00026302735,0.00048752967,0.55086136,0.022664644,0.3573368,0.06693388,0.00008964242],"about_ca_topic_score_codex":0.0062778164,"about_ca_topic_score_gemma":0.0062235096,"teacher_disagreement_score":0.015178596,"about_ca_system_score_codex":0.002459456,"about_ca_system_score_gemma":0.003377972,"threshold_uncertainty_score":0.050777435},"labels":[],"label_agreement":null},{"id":"W4361295185","doi":"10.22329/il.v43i1.7639","title":"The Broad Reach of Multivariable Thinking","year":2023,"lang":"en","type":"article","venue":"Informal Logic","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Task (project management); Multivariable calculus; Causal reasoning; Simple (philosophy); Sample (material); Psychology; Cognitive psychology; Social psychology; Cognition; Epistemology; Psychiatry","score_opus":0.020350997013837597,"score_gpt":0.2876073364846006,"score_spread":0.267256339470763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4361295185","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.672883,0.0038675289,0.22862324,0.015094107,0.00013627314,0.00024952277,0.00024300936,0.00045430724,0.07844906],"genre_scores_gemma":[0.9857041,0.00028921594,0.012907428,0.00031695317,0.00004101109,0.0000455822,0.000038122347,0.000027761724,0.0006298808],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.981682,0.010818915,0.0011150215,0.0017951133,0.003982299,0.0006066958],"domain_scores_gemma":[0.8425685,0.12914672,0.009462136,0.01168014,0.0058548534,0.0012876207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024100712,0.00053150865,0.0004528167,0.003071831,0.0012104991,0.0039826785,0.0012174607,0.0014511406,0.0061901323],"category_scores_gemma":[0.12556249,0.00048263185,0.00058169814,0.0013183533,0.0071041635,0.007293737,0.0054914514,0.0024772214,0.00041342626],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029812506,0.00018367953,0.146511,0.001492824,0.00028245206,0.0016458553,0.11565681,0.003190744,0.0074192327,0.35498592,0.0038110383,0.36452237],"study_design_scores_gemma":[0.00006338198,0.00016218159,0.049092054,0.0009608455,0.00011805649,0.0015746973,0.020402042,0.011809364,0.0032025361,0.8754167,0.037090845,0.00010738518],"about_ca_topic_score_codex":0.0012921066,"about_ca_topic_score_gemma":0.0015328432,"teacher_disagreement_score":0.024100712,"about_ca_system_score_codex":0.0012198961,"about_ca_system_score_gemma":0.0016576414,"threshold_uncertainty_score":0.12745821},"labels":[],"label_agreement":null},{"id":"W4365451534","doi":"10.1145/3592604","title":"Semi-Supervised Lexicon-Aware Embedding for News Article Time Estimation","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Asian and Low-Resource Language Information Processing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brandon University","funders":"","keywords":"Computer science; Artificial intelligence; Lexicon; Classifier (UML); Natural language processing; Stop words; WordNet; Machine learning","score_opus":0.008482545231234899,"score_gpt":0.2775019277698858,"score_spread":0.26901938253865093,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4365451534","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10436615,0.0011315994,0.8849159,0.00023751688,0.00014308761,0.00009971507,0.0006960435,0.006000882,0.002409088],"genre_scores_gemma":[0.80184245,0.0004614616,0.18666731,0.00019727078,0.00016956878,0.00021828598,0.004064748,0.00039944917,0.005979405],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994986,0.00013218945,0.00004653546,0.00015358687,0.000111333364,0.000057673566],"domain_scores_gemma":[0.9982822,0.0007222213,0.00023305907,0.0002131442,0.0004956228,0.00005378819],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00075264095,0.0009812256,0.0008390005,0.0011806941,0.00023938199,0.00068094593,0.0010459372,0.00062806555,0.0012990407],"category_scores_gemma":[0.0036653797,0.00041440222,0.0005864601,0.000918639,0.0003938693,0.0019709058,0.00076038437,0.00094522006,0.0015045876],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058166956,0.0004323059,0.0039687515,0.00026893523,0.00017927255,0.00021431166,0.00022583727,0.21333718,0.03447475,0.004160315,0.007841494,0.73431516],"study_design_scores_gemma":[0.000008515519,0.000047753456,0.00048433585,0.000007001823,0.000012535088,0.000039884704,0.00002205371,0.9920656,0.004489966,0.0021097802,0.000700831,0.00001171257],"about_ca_topic_score_codex":0.0037312966,"about_ca_topic_score_gemma":0.006727579,"teacher_disagreement_score":0.0037312966,"about_ca_system_score_codex":0.00056818774,"about_ca_system_score_gemma":0.0006847917,"threshold_uncertainty_score":0.007419169},"labels":[],"label_agreement":null},{"id":"W4366136173","doi":"10.2139/ssrn.4421954","title":"A First Look at Information Highlighting in Stack Overflow Answers","year":2023,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; University of Manitoba","funders":"","keywords":"Stack (abstract data type); Computer science; Data science; World Wide Web; Programming language","score_opus":0.010915847275495192,"score_gpt":0.2544394522229137,"score_spread":0.2435236049474185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4366136173","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24506961,0.017197473,0.47224692,0.026024563,0.0031414875,0.00063721865,0.0078872815,0.03036364,0.19743183],"genre_scores_gemma":[0.7581055,0.0041245827,0.15728544,0.004336721,0.0019705442,0.00013027937,0.003587743,0.005848036,0.06461108],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99618584,0.000984816,0.0002693901,0.00037891965,0.0016642782,0.0005168855],"domain_scores_gemma":[0.97614247,0.016195854,0.0014852325,0.0016701672,0.004071967,0.00043428704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029168413,0.00073119916,0.00084470754,0.0055777,0.0025630826,0.004873452,0.0012208767,0.0025130014,0.030976146],"category_scores_gemma":[0.033148363,0.0005324878,0.00065941474,0.005475269,0.0018793904,0.012381703,0.0027935358,0.0023183853,0.0058534667],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019256287,0.0002631346,0.01494463,0.0023730225,0.00010457453,0.0050757127,0.019945774,0.0038061189,0.052621808,0.22549497,0.123722374,0.54972225],"study_design_scores_gemma":[0.00009907197,0.0006301409,0.013020668,0.0017548172,0.00024256103,0.0054584797,0.010762022,0.0327257,0.0808132,0.1959128,0.6582339,0.00034680378],"about_ca_topic_score_codex":0.0025082547,"about_ca_topic_score_gemma":0.0028340064,"teacher_disagreement_score":0.030976146,"about_ca_system_score_codex":0.0011802426,"about_ca_system_score_gemma":0.0011894766,"threshold_uncertainty_score":0.103625536},"labels":[],"label_agreement":null},{"id":"W4366431549","doi":"10.23977/jaip.2023.060110","title":"AI Application to Generate an Expected Picture Using Keywords with Stable Diffusion","year":2023,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Practice","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Noise (video); Image (mathematics); Artificial intelligence; Generator (circuit theory); Field (mathematics); Painting; Creativity; Diffusion; Process (computing); Computer vision; Visual arts; Law; Mathematics; Art; Power (physics); Programming language","score_opus":0.045394645205839315,"score_gpt":0.3835652351684803,"score_spread":0.338170589962641,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4366431549","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034509137,0.0006855475,0.9244999,0.0011233648,0.00028989272,0.00049446436,0.0005527133,0.008991453,0.028853469],"genre_scores_gemma":[0.39592054,0.0008767395,0.5664105,0.00025406343,0.000094588315,0.00033812507,0.0010088222,0.00086521555,0.03423137],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995503,0.000091070324,0.000042714993,0.00009856417,0.00019048507,0.000026886419],"domain_scores_gemma":[0.9992918,0.0003143021,0.000047303416,0.00010539254,0.00019685016,0.00004444651],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006340397,0.0004597825,0.00033043267,0.0010101752,0.00047791103,0.0013259395,0.0007789559,0.0007626245,0.0131338],"category_scores_gemma":[0.0036274304,0.00017014718,0.0005890247,0.0008434147,0.00038560457,0.0021636146,0.0007463226,0.00043526798,0.0033517056],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005575608,0.00021440031,0.0023064346,0.0013473092,0.00007464955,0.0015809062,0.0015947814,0.03633319,0.113488644,0.11678375,0.028324747,0.6973936],"study_design_scores_gemma":[0.00010483917,0.00027993368,0.0012401546,0.00010019223,0.00006423154,0.0016372616,0.00045593028,0.7584859,0.08290822,0.062566414,0.09207853,0.000078315396],"about_ca_topic_score_codex":0.0012079121,"about_ca_topic_score_gemma":0.0008892489,"teacher_disagreement_score":0.0131338,"about_ca_system_score_codex":0.00049991685,"about_ca_system_score_gemma":0.0004553291,"threshold_uncertainty_score":0.04393691},"labels":[],"label_agreement":null},{"id":"W4367293423","doi":"10.5220/0011826700003467","title":"Novel Topic Models for Content Based Recommender Systems","year":2023,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Recommender system; Computer science; Content (measure theory); Information retrieval; Mathematics","score_opus":0.2072451708935658,"score_gpt":0.32651371135770835,"score_spread":0.11926854046414254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367293423","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013974857,0.0017184016,0.9794119,0.00093598745,0.0002485536,0.00013024984,0.00082832563,0.0009244324,0.0018273066],"genre_scores_gemma":[0.4851239,0.003826521,0.484314,0.00065082713,0.0017786237,0.0009210805,0.0045926925,0.0006359387,0.018156487],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966222,0.0015040142,0.00024037232,0.0006605566,0.00073833286,0.00023450534],"domain_scores_gemma":[0.9890315,0.0077655576,0.0004455852,0.0010710282,0.0013698373,0.0003165528],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042315857,0.0012867568,0.002190588,0.002460191,0.0012948834,0.0036714345,0.0028245756,0.0024925142,0.004048535],"category_scores_gemma":[0.01797966,0.0010930804,0.001890862,0.003584226,0.00076620805,0.0062168543,0.0018390012,0.0032826895,0.0031262168],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010688874,0.00075214973,0.0070507894,0.00091695675,0.00085624785,0.00036116238,0.0012939743,0.30546674,0.012307375,0.19776237,0.039765213,0.43239814],"study_design_scores_gemma":[0.000027389233,0.000040943356,0.00031886128,0.000021360214,0.00006752337,0.0000594603,0.00003523592,0.9615578,0.00056029344,0.034475837,0.0028121085,0.000023218288],"about_ca_topic_score_codex":0.0064753382,"about_ca_topic_score_gemma":0.010589732,"teacher_disagreement_score":0.0064753382,"about_ca_system_score_codex":0.0014949342,"about_ca_system_score_gemma":0.0014361822,"threshold_uncertainty_score":0.02237904},"labels":[],"label_agreement":null},{"id":"W4378364165","doi":"10.1145/3589462.3589463","title":"ConfSys - An Intelligent Conference Management System","year":2023,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Process (computing); Automation; Relevance (law); Metadata; Salient; Matching (statistics); Quality (philosophy); Software engineering; Information retrieval; World Wide Web; Data science; Artificial intelligence; Engineering; Programming language","score_opus":0.04069419251764079,"score_gpt":0.30958856042868116,"score_spread":0.2688943679110404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378364165","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022790113,0.0019249645,0.17461288,0.0019951533,0.001384182,0.001566012,0.029239343,0.70850456,0.057982843],"genre_scores_gemma":[0.39173317,0.002448366,0.33696333,0.0030664098,0.0034775885,0.0031964583,0.14454,0.024713418,0.089861214],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.996997,0.00069709774,0.0004312941,0.00061134057,0.0010713574,0.0001919289],"domain_scores_gemma":[0.99201787,0.0016264698,0.0007757075,0.001990105,0.0023147925,0.0012752156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044565317,0.0010813061,0.0009868456,0.0055260723,0.0012313455,0.0044239173,0.0025903657,0.0010769499,0.027484022],"category_scores_gemma":[0.009616112,0.0006488517,0.00068522553,0.0032703949,0.00044163587,0.003850898,0.0028908988,0.0012476364,0.018248288],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014694232,0.0003168906,0.005232195,0.00081863004,0.00018065287,0.0004553261,0.0005223622,0.0040242025,0.009521595,0.007631411,0.6245419,0.34528536],"study_design_scores_gemma":[0.00057774276,0.0005175512,0.008288072,0.00020182873,0.00018476,0.00079121854,0.00044616198,0.10540191,0.030426264,0.012245602,0.84049827,0.00042061353],"about_ca_topic_score_codex":0.0022827683,"about_ca_topic_score_gemma":0.0015667703,"teacher_disagreement_score":0.027484022,"about_ca_system_score_codex":0.001392967,"about_ca_system_score_gemma":0.0022059313,"threshold_uncertainty_score":0.091943264},"labels":[],"label_agreement":null},{"id":"W4379184890","doi":"10.1155/2023/2570824","title":"A Function Area Division Approach for Autonomous Transportation System Based on Text Similarity","year":2023,"lang":"en","type":"article","venue":"Journal of Advanced Transportation","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Shenzhen Science and Technology Innovation Program; National Key Research and Development Program of China","keywords":"Similarity (geometry); Function (biology); Cluster analysis; Computer science; Context (archaeology); Hierarchical clustering; Data mining; Artificial intelligence; Geography; Image (mathematics)","score_opus":0.017789884034501297,"score_gpt":0.2629471552590599,"score_spread":0.24515727122455858,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379184890","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03232859,0.0002798121,0.9618608,0.00015254831,0.000043761123,0.00027779108,0.0003550007,0.0011690743,0.0035327084],"genre_scores_gemma":[0.34830806,0.0001890736,0.6455123,0.00008624169,0.00007674702,0.0004849689,0.0014505329,0.00022086337,0.0036713418],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981053,0.0003671046,0.0001868059,0.00063678005,0.00058412156,0.00011997267],"domain_scores_gemma":[0.9977399,0.0007275691,0.0003378134,0.00027357304,0.0008193798,0.00010175794],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011369181,0.00088636164,0.0007835344,0.006420417,0.0013243462,0.0017253722,0.001472048,0.0010239425,0.0029310333],"category_scores_gemma":[0.005161178,0.00027773954,0.0010461236,0.004290878,0.00094828295,0.0028851358,0.0015817492,0.00092799845,0.0013561344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004090991,0.0003100338,0.007883904,0.0004990906,0.00013975598,0.00041944615,0.003297996,0.07152063,0.026166162,0.038679704,0.0065260576,0.8441482],"study_design_scores_gemma":[0.000051527,0.00024364493,0.0067742825,0.00007209841,0.00013437207,0.00039803106,0.0018295115,0.9075568,0.01787371,0.04684537,0.018139971,0.000080632584],"about_ca_topic_score_codex":0.0050418945,"about_ca_topic_score_gemma":0.0054988884,"teacher_disagreement_score":0.006420417,"about_ca_system_score_codex":0.001258652,"about_ca_system_score_gemma":0.0013126895,"threshold_uncertainty_score":0.010025084},"labels":[],"label_agreement":null},{"id":"W4379984080","doi":"10.1109/aeis59450.2022.00030","title":"Behavioral Mapping, Using NLP to Predict Individual Behavior : Focusing on Towards/Away Behavior","year":2022,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Mitacs","keywords":"Artificial intelligence; Computer science; Natural language processing","score_opus":0.08143062791456686,"score_gpt":0.3463467068363596,"score_spread":0.26491607892179275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379984080","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76338863,0.00050316297,0.20969227,0.00089096377,0.00014480545,0.0008658954,0.009779561,0.0031370823,0.01159768],"genre_scores_gemma":[0.80394906,0.0002474616,0.18550296,0.00011673739,0.000051519208,0.0005081259,0.0058694743,0.00007846846,0.0036760792],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991715,0.00039354802,0.00007945117,0.00017196102,0.00013888943,0.00004466402],"domain_scores_gemma":[0.9919412,0.006325805,0.0005172026,0.00034415178,0.00067980826,0.00019175767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015908995,0.00071456376,0.00041347335,0.0031735757,0.00046858395,0.0010555862,0.00041374518,0.000609677,0.0021028747],"category_scores_gemma":[0.0076730377,0.00012925733,0.00049698557,0.0016734567,0.00024294476,0.0010106879,0.0005090737,0.0006198611,0.0014013369],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055974175,0.0013055259,0.41263303,0.00067442446,0.0002561127,0.00028140782,0.002117604,0.015277595,0.011110281,0.0016565626,0.0059909066,0.54813683],"study_design_scores_gemma":[0.00006477193,0.0007913995,0.45118433,0.00020061476,0.0002727247,0.0005929039,0.0035403403,0.5043506,0.018799616,0.009177949,0.010873606,0.00015114325],"about_ca_topic_score_codex":0.0053656395,"about_ca_topic_score_gemma":0.007233582,"teacher_disagreement_score":0.0053656395,"about_ca_system_score_codex":0.00041286266,"about_ca_system_score_gemma":0.0005635268,"threshold_uncertainty_score":0.010668814},"labels":[],"label_agreement":null},{"id":"W4380446958","doi":"10.1177/07356331231178873","title":"Predicting the Persuasiveness of Influence Strategies From Student Online Learning Behaviour Using Machine Learning Methods","year":2023,"lang":"en","type":"article","venue":"Journal of Educational Computing Research","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Adaptation (eye); Artificial intelligence; Machine learning; Psychology","score_opus":0.11991158000098773,"score_gpt":0.5442643351950351,"score_spread":0.4243527551940474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380446958","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9505783,0.0002128322,0.04618061,0.00014441009,0.000023581471,0.00024411782,0.00039847946,0.00053867453,0.0016789866],"genre_scores_gemma":[0.98190814,0.00006363929,0.017052436,0.000021663598,0.000013210929,0.00013159671,0.0003236848,0.0000140731045,0.00047162222],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9977717,0.0010007052,0.00023023065,0.00038857802,0.00047556637,0.00013320448],"domain_scores_gemma":[0.96036404,0.033941884,0.0022368922,0.00094996905,0.0020462933,0.00046094126],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035275363,0.0008643978,0.00078647456,0.003342656,0.0002637058,0.0016029724,0.00053665275,0.0009223669,0.0012719859],"category_scores_gemma":[0.028516423,0.00031087623,0.00080266845,0.0012747579,0.00030781614,0.0011399525,0.0004270379,0.0011303255,0.0006303876],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094426144,0.003665004,0.5546659,0.000486643,0.0005660938,0.00015935491,0.0015763504,0.06740814,0.006703934,0.00062615436,0.0010978257,0.36210027],"study_design_scores_gemma":[0.000036542853,0.0007385025,0.18112767,0.00006688546,0.000119061464,0.00009527379,0.0003525892,0.80957216,0.0058641373,0.0013273478,0.0006295442,0.000070246744],"about_ca_topic_score_codex":0.003602665,"about_ca_topic_score_gemma":0.0040150178,"teacher_disagreement_score":0.003602665,"about_ca_system_score_codex":0.0006132997,"about_ca_system_score_gemma":0.0005026133,"threshold_uncertainty_score":0.018655598},"labels":[],"label_agreement":null},{"id":"W4380761095","doi":"10.2139/ssrn.4477833","title":"Demand Forecasting of New Products Using Language Models on the Product Descriptions","year":2023,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Product (mathematics); Demand forecasting; Econometrics; Computer science; Economics; Mathematics; Operations management","score_opus":0.10737581900762051,"score_gpt":0.31181763415626995,"score_spread":0.20444181514864945,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380761095","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5841273,0.001266337,0.40107435,0.0012839468,0.00016177456,0.00013311006,0.0071761943,0.0011667468,0.0036102342],"genre_scores_gemma":[0.9496714,0.0006433349,0.04113876,0.00006267026,0.00008652216,0.00009239743,0.0050578876,0.00008897714,0.0031580457],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994425,0.0002356485,0.000036973266,0.00012747648,0.00009404816,0.00006343797],"domain_scores_gemma":[0.99466324,0.004276051,0.00040590318,0.00020007083,0.0003678076,0.000086948385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011734567,0.00061051315,0.00076570554,0.0017049033,0.00023176354,0.0012973815,0.0008347036,0.0005846672,0.0024925177],"category_scores_gemma":[0.005343119,0.00045486644,0.0011924215,0.0025375807,0.00029886744,0.0025218453,0.00041142214,0.0012403104,0.001158004],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010906202,0.00050145015,0.018039104,0.00034883755,0.00021937929,0.00038190762,0.0002560717,0.76766384,0.006814644,0.018180436,0.00554514,0.18095863],"study_design_scores_gemma":[0.0000084711955,0.000018472485,0.0007800724,0.00000476313,0.000010957461,0.0000092392065,0.000011861,0.9962894,0.00028403598,0.002375722,0.0002000067,0.000007032666],"about_ca_topic_score_codex":0.018182464,"about_ca_topic_score_gemma":0.016550407,"teacher_disagreement_score":0.018182464,"about_ca_system_score_codex":0.0011077637,"about_ca_system_score_gemma":0.00074216863,"threshold_uncertainty_score":0.036153257},"labels":[],"label_agreement":null},{"id":"W4383818705","doi":"10.33423/jabe.v25i3.6208","title":"Conceptual Meta-Models: An Example Correlating Anthony’s Triangle, Simon’s Structure, and Stevens’ Scale of Measurement","year":2023,"lang":"en","type":"article","venue":"Journal of Applied Business and Economics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Acknowledgement; Simple (philosophy); Brainstorming; Scale (ratio); Computer science; Craft; Epistemology; Mathematical economics; Core (optical fiber); Econometrics; Artificial intelligence; Data science; Psychology; Mathematics; Philosophy; Visual arts; Art","score_opus":0.11024792538248765,"score_gpt":0.253274796042182,"score_spread":0.14302687065969436,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383818705","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032921836,0.0059865415,0.54257745,0.084844366,0.00084733055,0.00026423144,0.0002813767,0.0005593688,0.33171746],"genre_scores_gemma":[0.7791585,0.00259546,0.19865741,0.0036937573,0.00033831992,0.00031132792,0.0002052237,0.00043410365,0.014605932],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98446333,0.010824211,0.00057138264,0.0007420421,0.0030288107,0.000370233],"domain_scores_gemma":[0.980717,0.012887051,0.0008881247,0.0027353598,0.002171923,0.0006004692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0138596725,0.0010042704,0.00067729264,0.0048773726,0.0052056946,0.007985581,0.002229433,0.004254322,0.004773915],"category_scores_gemma":[0.02355796,0.00059757056,0.0012189368,0.004992703,0.022787282,0.019983275,0.005828585,0.0061391597,0.0009300356],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000007928052,0.000009446073,0.00013848588,0.000027693028,0.0000037255854,0.000070632705,0.0048831343,0.00014345697,0.00006831543,0.9885048,0.0022522777,0.0038901507],"study_design_scores_gemma":[0.00001266746,0.00001773604,0.00023145495,0.0001143482,0.000010122088,0.0003542521,0.0039418386,0.0037848903,0.00021182236,0.9122723,0.07902626,0.000022415134],"about_ca_topic_score_codex":0.0059098112,"about_ca_topic_score_gemma":0.008752038,"teacher_disagreement_score":0.0138596725,"about_ca_system_score_codex":0.0074249213,"about_ca_system_score_gemma":0.0034151217,"threshold_uncertainty_score":0.07329786},"labels":[],"label_agreement":null},{"id":"W4385567143","doi":"10.18653/v1/2022.finnlp-1.10","title":"A Taxonomical NLP Blueprint to Support Financial Decision Making through Information-Centred Interactions","year":2022,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Blueprint; Usability; Taxonomy (biology); Visualization; Artificial intelligence; Data science; Software; Natural language processing; Human–computer interaction","score_opus":0.026220031128080044,"score_gpt":0.3076132401009309,"score_spread":0.28139320897285086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385567143","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014114087,0.00038958053,0.9189242,0.015266826,0.00014379251,0.0012278581,0.00049182784,0.0022782043,0.04716359],"genre_scores_gemma":[0.09430571,0.0002704305,0.8970713,0.0010565022,0.000038811093,0.0016506226,0.0004743479,0.0002469553,0.0048852214],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9785831,0.016733352,0.001126321,0.0011413543,0.0020589635,0.00035693098],"domain_scores_gemma":[0.9676725,0.021026049,0.0014381944,0.0055618,0.0033002174,0.0010013008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017495077,0.0009726473,0.00041470377,0.0052581453,0.003825127,0.007646142,0.0025648372,0.0025683136,0.00901785],"category_scores_gemma":[0.030614553,0.0007009375,0.0005908602,0.0037607318,0.009778819,0.013874564,0.0054489337,0.0032833326,0.0023665754],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012489138,0.00032808047,0.0025217768,0.0011678268,0.000022680295,0.00050703273,0.11801841,0.0030733824,0.015429769,0.553675,0.017399736,0.28773144],"study_design_scores_gemma":[0.000062161664,0.00020045109,0.0028941932,0.001346631,0.00002339626,0.001080176,0.03977702,0.020999266,0.0051767123,0.4431604,0.48518428,0.000095316325],"about_ca_topic_score_codex":0.003657458,"about_ca_topic_score_gemma":0.00607762,"teacher_disagreement_score":0.017495077,"about_ca_system_score_codex":0.0032457614,"about_ca_system_score_gemma":0.0058464264,"threshold_uncertainty_score":0.09252393},"labels":[],"label_agreement":null},{"id":"W4385834249","doi":"10.1109/access.2023.3305260","title":"Summarizing Students’ Free Responses for an Introductory Algebra-Based Physics Course Survey Using Cluster and Sentiment Analysis","year":2023,"lang":"en","type":"article","venue":"IEEE Access","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"University of Toronto","keywords":"Sentiment analysis; Likert scale; Computer science; Valence (chemistry); Set (abstract data type); Cluster grouping; Mathematics education; Macro; Natural language processing; Text messaging; Artificial intelligence; Information retrieval; Psychology; Statistics; World Wide Web; Mathematics","score_opus":0.10250271593371403,"score_gpt":0.42651022883468137,"score_spread":0.32400751290096735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385834249","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94183373,0.00010386593,0.04132191,0.00027010875,0.00011238481,0.0010077333,0.011319801,0.0012540189,0.0027763923],"genre_scores_gemma":[0.91141343,0.00012024071,0.06500967,0.00015523547,0.00014863415,0.002439778,0.017592276,0.00021207125,0.0029086184],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99612963,0.0015250521,0.00043786204,0.0004783476,0.0012252467,0.0002039209],"domain_scores_gemma":[0.97806656,0.011212881,0.0024812059,0.0017011221,0.005919755,0.000618384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032857382,0.0006883158,0.00062116014,0.004381629,0.00045651407,0.00087119796,0.000351636,0.00042050707,0.0025519985],"category_scores_gemma":[0.018268233,0.00013757891,0.00044714892,0.0034171313,0.00022797807,0.00057280064,0.00084424886,0.0004614985,0.0013984751],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013323586,0.0011688784,0.33389848,0.0014186198,0.0004878715,0.00023217508,0.010714837,0.0056418707,0.055824544,0.0010540256,0.04260071,0.5456256],"study_design_scores_gemma":[0.00008253019,0.0012995163,0.88307047,0.00012496568,0.00018563839,0.00017159808,0.011203545,0.051764566,0.026079101,0.0021550362,0.023588628,0.00027433794],"about_ca_topic_score_codex":0.0010960475,"about_ca_topic_score_gemma":0.0024701594,"teacher_disagreement_score":0.004381629,"about_ca_system_score_codex":0.00042849878,"about_ca_system_score_gemma":0.00040606296,"threshold_uncertainty_score":0.01737684},"labels":[],"label_agreement":null},{"id":"W4386454333","doi":"10.2139/ssrn.4559513","title":"An Exploration of the Effects of Message Framing on Plant-based Meat Alternatives","year":2023,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of the Fraser Valley","funders":"","keywords":"Framing (construction); Business; Advertising; Engineering; Civil engineering","score_opus":0.013047608597176692,"score_gpt":0.28604436140732004,"score_spread":0.27299675281014335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386454333","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97359586,0.00015205037,0.0019586957,0.00032081825,0.000038473398,0.00006823499,0.00012424767,0.000019470019,0.023722157],"genre_scores_gemma":[0.9966132,0.000074262854,0.0012444779,0.00008424165,0.000022031965,0.000049161386,0.000062739775,0.000022219847,0.0018277244],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99727947,0.0019907788,0.000069443595,0.00022670659,0.0002950358,0.00013861891],"domain_scores_gemma":[0.8224715,0.1686942,0.0047995676,0.001576141,0.001640589,0.00081802317],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050089485,0.000514697,0.00044137033,0.0005464074,0.0006149503,0.0028856765,0.0007130655,0.0010693474,0.019071821],"category_scores_gemma":[0.06976803,0.00029286926,0.0004124673,0.00056074123,0.0008521204,0.0021225207,0.0010536752,0.0019987216,0.0008390833],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.1215114,0.020855648,0.1881466,0.004503849,0.0013703482,0.0026960515,0.060770456,0.034842342,0.13992472,0.13072981,0.0044936594,0.29015505],"study_design_scores_gemma":[0.00506416,0.023048168,0.64109766,0.0007571061,0.004296106,0.0007703353,0.039387167,0.09843039,0.028830824,0.13666011,0.021212744,0.0004452047],"about_ca_topic_score_codex":0.0023156758,"about_ca_topic_score_gemma":0.002096189,"teacher_disagreement_score":0.019071821,"about_ca_system_score_codex":0.00084158836,"about_ca_system_score_gemma":0.0005181792,"threshold_uncertainty_score":0.06380159},"labels":[],"label_agreement":null},{"id":"W4386565867","doi":"10.1016/j.infsof.2023.107321","title":"A large-scale exploratory study of android sports apps in the google play store","year":2023,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; World Wide Web; Android (operating system); Empirical research; Sentiment analysis; App store; Set (abstract data type); Data science; Internet privacy; Artificial intelligence","score_opus":0.008951972813567967,"score_gpt":0.25349437146295273,"score_spread":0.24454239864938476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386565867","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99820256,0.000055429246,0.00011615118,0.00009412562,0.0000061115297,0.00010928577,0.00054585905,0.00001380338,0.0008567829],"genre_scores_gemma":[0.9947397,0.00017542754,0.0007406192,0.00028170648,0.00002392341,0.00023841087,0.0012150173,0.000043815082,0.0025413847],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9987016,0.00038234988,0.00008094999,0.0001756631,0.0004419102,0.00021755368],"domain_scores_gemma":[0.9923735,0.004194774,0.0011403578,0.00033509597,0.0012057825,0.0007503489],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012261425,0.00045067244,0.0005234912,0.0025429416,0.0022501487,0.0024931119,0.00077916525,0.00093949446,0.0020598627],"category_scores_gemma":[0.007844859,0.00046802056,0.0004067328,0.0022037243,0.0014878035,0.0031226287,0.0021135954,0.0011927844,0.0010225177],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010247302,0.0033927364,0.57222,0.00090958294,0.00023341282,0.0038291167,0.3646184,0.00020959998,0.009080113,0.0010853214,0.0075169206,0.03588004],"study_design_scores_gemma":[0.000041040552,0.00071393803,0.714232,0.00017984131,0.000081555394,0.00076313445,0.27335775,0.0008840205,0.00097205455,0.00021462185,0.008468858,0.00009108997],"about_ca_topic_score_codex":0.021213347,"about_ca_topic_score_gemma":0.05462163,"teacher_disagreement_score":0.021213347,"about_ca_system_score_codex":0.00085037807,"about_ca_system_score_gemma":0.0013395955,"threshold_uncertainty_score":0.042179763},"labels":[],"label_agreement":null},{"id":"W4386645575","doi":"10.1007/978-3-031-41456-5_16","title":"An Abstractive Automatic Summarization Approach Based on a Text Comprehension Model of Cognitive Psychology","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Université du Québec à Trois-Rivières","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Comprehension; Cognition; Artificial intelligence; Cognitive model; Cognitive science; Information retrieval; Programming language; Psychology","score_opus":0.035853468901795464,"score_gpt":0.31732284320819737,"score_spread":0.2814693743064019,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386645575","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0051810676,0.00015888386,0.98902136,0.00025798817,0.00007829579,0.00015353858,0.00019283763,0.0026157412,0.002340243],"genre_scores_gemma":[0.10312505,0.0002497388,0.8867172,0.00018471923,0.0002058659,0.00040310028,0.0012525913,0.00048137084,0.0073803933],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990108,0.00042779138,0.00007946507,0.00024860617,0.00019159951,0.000041682706],"domain_scores_gemma":[0.99725854,0.0013536714,0.00018003849,0.0003219275,0.0008077681,0.00007807471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012291735,0.0011716188,0.00089456025,0.0014661456,0.0006153358,0.0025192366,0.0017522691,0.0009211733,0.009583782],"category_scores_gemma":[0.0052452693,0.00044211012,0.0012072553,0.0012079196,0.00061800913,0.0042008413,0.0012379552,0.0014695452,0.0034664827],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003617926,0.00039760408,0.00071389315,0.00094342884,0.0001740319,0.00023198774,0.0018905195,0.012519469,0.07969574,0.0795317,0.018564584,0.8049753],"study_design_scores_gemma":[0.00011501017,0.00050182303,0.001907434,0.00011188576,0.00030575835,0.0002786525,0.0007194987,0.74408853,0.045418356,0.17795923,0.028485386,0.00010837289],"about_ca_topic_score_codex":0.000915401,"about_ca_topic_score_gemma":0.0012155015,"teacher_disagreement_score":0.009583782,"about_ca_system_score_codex":0.00045906493,"about_ca_system_score_gemma":0.0007741591,"threshold_uncertainty_score":0.03206092},"labels":[],"label_agreement":null},{"id":"W4388422667","doi":"10.18280/isi.280530","title":"Leveraging Latent Dirichlet Allocation for Feature Extraction in User Comments: Enhancements to User-Centered Design in Indonesian Financial Technology","year":2023,"lang":"en","type":"article","venue":"Ingénierie des systèmes d information","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Latent Dirichlet allocation; Indonesian; Computer science; Dirichlet distribution; Topic model; Data mining; Information retrieval; Mathematics","score_opus":0.02612105426512728,"score_gpt":0.29246254596887966,"score_spread":0.2663414917037524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388422667","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08830635,0.00027817892,0.90578884,0.00043479222,0.000074769516,0.0010168295,0.00051909935,0.0020754272,0.0015057491],"genre_scores_gemma":[0.4507614,0.00018306077,0.5432078,0.00018400802,0.00008549745,0.0019306741,0.0012345312,0.00021453826,0.0021984477],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98474944,0.010333832,0.0009574388,0.0020724763,0.0015376667,0.00034903153],"domain_scores_gemma":[0.9736767,0.019924104,0.0013549051,0.00133918,0.0033188614,0.0003863905],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012486936,0.0009378662,0.0010501422,0.0027386684,0.00079305284,0.0026385605,0.0010877594,0.0010248095,0.0017359193],"category_scores_gemma":[0.029474769,0.00046170576,0.001176325,0.0017973027,0.00095208565,0.0028540252,0.0018317447,0.001282259,0.0012271601],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002689916,0.0011186722,0.024956267,0.0013130517,0.00023900114,0.00040169858,0.014082336,0.021091808,0.039923165,0.007238785,0.004123877,0.88282144],"study_design_scores_gemma":[0.00021035336,0.0010264845,0.025793327,0.00023212536,0.00022227026,0.0004587114,0.0054017883,0.8919045,0.036254894,0.022404693,0.015797574,0.00029325378],"about_ca_topic_score_codex":0.0017556689,"about_ca_topic_score_gemma":0.0029240977,"teacher_disagreement_score":0.012486936,"about_ca_system_score_codex":0.00093385676,"about_ca_system_score_gemma":0.0014598576,"threshold_uncertainty_score":0.06603801},"labels":[],"label_agreement":null},{"id":"W4388481904","doi":"10.48550/arxiv.2311.02985","title":"Towards a Transformer-Based Reverse Dictionary Model for Quality Estimation of Definitions","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Transformer; Computer science; Artificial intelligence; Engineering; Electrical engineering; Voltage","score_opus":0.28084115684796945,"score_gpt":0.2807720832417779,"score_spread":0.00006907360619157199,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388481904","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003906715,0.000104716506,0.9950299,0.00014245586,0.000016055443,0.00003674942,0.0000892075,0.0002792786,0.00039485202],"genre_scores_gemma":[0.43348256,0.0006305284,0.5593873,0.00037370538,0.00013114153,0.0003426979,0.0013253082,0.00047632557,0.0038504538],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99579394,0.0018953255,0.0003159156,0.0009743723,0.0007635893,0.00025679983],"domain_scores_gemma":[0.98306674,0.010995244,0.001190427,0.0019718492,0.0023429175,0.0004328301],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0066199596,0.0011907332,0.0015495509,0.0026654822,0.00045027223,0.0034218438,0.0031982225,0.0018372613,0.004734564],"category_scores_gemma":[0.035075255,0.0008612022,0.0018157449,0.002314623,0.0020637724,0.00838238,0.0031872126,0.0038752086,0.0018557009],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078247674,0.00031192676,0.006682551,0.0005745112,0.00029554384,0.00023046762,0.00097836,0.28383914,0.008921698,0.3354134,0.006511926,0.35545802],"study_design_scores_gemma":[0.000018354236,0.00005839382,0.0002685301,0.000021930118,0.000022583848,0.000057400735,0.0000373569,0.91222006,0.0007908527,0.08549584,0.0009916733,0.000017046988],"about_ca_topic_score_codex":0.0046959314,"about_ca_topic_score_gemma":0.005210008,"teacher_disagreement_score":0.0066199596,"about_ca_system_score_codex":0.0013961942,"about_ca_system_score_gemma":0.001418187,"threshold_uncertainty_score":0.0350101},"labels":[],"label_agreement":null},{"id":"W4389289483","doi":"10.33423/jabe.v25i6.6570","title":"Big Data Measures of Environmental Concern","year":2023,"lang":"en","type":"article","venue":"Journal of Applied Business and Economics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Snapshot (computer storage); Big data; Environmental data; Survey data collection; Environmental pollution; Environmental resource management; Scale (ratio); Data science; Environmental planning; Environmental science; Business; Computer science; Geography; Environmental protection; Data mining; Statistics; Political science; Database; Cartography; Mathematics","score_opus":0.07659842379509464,"score_gpt":0.24882667005392708,"score_spread":0.17222824625883243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389289483","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5690654,0.0061390838,0.052925568,0.0075323074,0.0004772347,0.000988627,0.3038472,0.0008020398,0.05822253],"genre_scores_gemma":[0.89077497,0.0014895401,0.021259844,0.00076206285,0.00034315372,0.00093888544,0.08192936,0.00009284135,0.0024091806],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99204564,0.0026632743,0.001185164,0.0008858248,0.0029684773,0.00025153704],"domain_scores_gemma":[0.90672344,0.0542398,0.02103687,0.0069903494,0.0090940185,0.0019155083],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003873313,0.00065419765,0.0005668984,0.010337285,0.0006517414,0.002219772,0.0007982008,0.00085948716,0.005355665],"category_scores_gemma":[0.049640868,0.00019216958,0.00068061985,0.02163667,0.00069231435,0.0040907282,0.0018221809,0.0009788997,0.0012046489],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027044056,0.00036785315,0.79528505,0.003547806,0.0011516458,0.0002767971,0.0036752196,0.008145798,0.0018514686,0.02873871,0.04104401,0.11564527],"study_design_scores_gemma":[0.00005187732,0.000315818,0.8149845,0.00082335074,0.0003850918,0.00053922145,0.0073050964,0.018898126,0.0033300181,0.04669881,0.10649039,0.00017776975],"about_ca_topic_score_codex":0.0034823022,"about_ca_topic_score_gemma":0.004123698,"teacher_disagreement_score":0.010337285,"about_ca_system_score_codex":0.0008407592,"about_ca_system_score_gemma":0.0007754992,"threshold_uncertainty_score":0.020484328},"labels":[],"label_agreement":null},{"id":"W4389519544","doi":"10.18653/v1/2023.newsum-1.12","title":"Analyzing Multi-Sentence Aggregation in Abstractive Summarization via the Shapley Value","year":2023,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research; Nvidia","keywords":"Automatic summarization; Shapley value; Computer science; Sentence; Value (mathematics); Natural language processing; Artificial intelligence; Mathematics; Game theory; Machine learning; Mathematical economics","score_opus":0.02542860140765619,"score_gpt":0.3057754320863267,"score_spread":0.2803468306786705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519544","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.115367845,0.0010684552,0.8745943,0.0009369505,0.00013749827,0.00018883072,0.000396987,0.00037148793,0.0069376547],"genre_scores_gemma":[0.83515173,0.00048837066,0.15840188,0.00022516001,0.00031860333,0.00023675505,0.0009671638,0.00017946766,0.0040309182],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9958028,0.0021889051,0.00024445972,0.0005994581,0.00090033613,0.00026415513],"domain_scores_gemma":[0.977815,0.017516723,0.0010417161,0.001211747,0.0019725228,0.00044233282],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006163673,0.0008606435,0.0019542258,0.0030205431,0.0013718251,0.0035815223,0.0014747924,0.0012748525,0.0038318557],"category_scores_gemma":[0.02689144,0.0005124105,0.0010956795,0.0032803987,0.0013621378,0.006756174,0.0019182413,0.0018503097,0.0004735364],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006300327,0.0004394092,0.004203546,0.000750754,0.000687686,0.00031192377,0.0011835836,0.29252493,0.006455566,0.4189008,0.009547352,0.26436448],"study_design_scores_gemma":[0.000027117358,0.00008766113,0.0007840359,0.000035184657,0.00007290374,0.00003413305,0.00010760273,0.6119676,0.0011290633,0.38467383,0.0010568955,0.0000238755],"about_ca_topic_score_codex":0.0014409539,"about_ca_topic_score_gemma":0.0017357499,"teacher_disagreement_score":0.006163673,"about_ca_system_score_codex":0.001490735,"about_ca_system_score_gemma":0.0014157986,"threshold_uncertainty_score":0.032597005},"labels":[],"label_agreement":null},{"id":"W4389519868","doi":"10.18653/v1/2023.findings-emnlp.199","title":"SimCKP: Simple Contrastive Learning of Keyphrase Representations","year":2023,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Artificial intelligence; Natural language processing; Phrase; Generator (circuit theory); Context (archaeology); Benchmark (surveying); Margin (machine learning); Simple (philosophy); Set (abstract data type); Machine learning; Power (physics)","score_opus":0.01824043602855783,"score_gpt":0.32997400725410264,"score_spread":0.3117335712255448,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519868","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021917704,0.00083030266,0.96712834,0.00015846515,0.00009174242,0.00019317683,0.00057447684,0.007375241,0.0017305514],"genre_scores_gemma":[0.30502036,0.00054936245,0.68173814,0.00039679528,0.00017842391,0.00040850404,0.0039899424,0.0006592836,0.007059179],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99910504,0.00018813513,0.00005541506,0.00036182042,0.00021356027,0.00007603805],"domain_scores_gemma":[0.9980567,0.0008985182,0.00017536696,0.00047563735,0.0003042509,0.00008954902],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014511601,0.0015311787,0.0011672649,0.001917711,0.00041877435,0.0012297941,0.00251898,0.0014397523,0.0036160182],"category_scores_gemma":[0.0049167587,0.000466259,0.0010398839,0.0016231794,0.00092851266,0.0031084039,0.0017030572,0.0022423645,0.003132155],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005296526,0.00036249918,0.0020076188,0.00038602485,0.00013591931,0.00022253922,0.0001234875,0.06424048,0.03374673,0.00831237,0.011730221,0.8782025],"study_design_scores_gemma":[0.000075834905,0.00029118411,0.00074952864,0.000023969687,0.000053209005,0.00020591645,0.0000453081,0.95722264,0.019681435,0.015443152,0.0061738803,0.00003401215],"about_ca_topic_score_codex":0.0014773142,"about_ca_topic_score_gemma":0.0035044577,"teacher_disagreement_score":0.0036160182,"about_ca_system_score_codex":0.0007658657,"about_ca_system_score_gemma":0.0011353862,"threshold_uncertainty_score":0.012096763},"labels":[],"label_agreement":null},{"id":"W4390475903","doi":"10.31436/iiumej.v25i1.1832","title":"Editorial","year":2024,"lang":"en","type":"editorial","venue":"IIUM Engineering Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Islam; Engineering; Malay; Library science; Management; Theology; Philosophy","score_opus":0.0028569979317511077,"score_gpt":0.24022953900490582,"score_spread":0.2373725410731547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390475903","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0002994118,0.014495844,0.0007931168,0.06157255,0.7315701,0.0002161691,0.0016686755,0.0009933073,0.1883908],"genre_scores_gemma":[0.004309347,0.018037882,0.0009805852,0.032287054,0.35634914,0.00018484693,0.002330844,0.00080575247,0.5847146],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9962619,0.00049868075,0.0003618778,0.0006905983,0.0018176232,0.00036927304],"domain_scores_gemma":[0.9812717,0.0018917526,0.00094743946,0.0012656202,0.0113379285,0.003285613],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0031333298,0.0018188719,0.0014517612,0.0033881632,0.0022743996,0.009668691,0.002421728,0.003637248,0.40240225],"category_scores_gemma":[0.023540199,0.00059510354,0.0011650154,0.0016160767,0.0010675367,0.0043482184,0.0022416837,0.0050022653,0.3173542],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014948847,0.0000042743636,0.000032170647,0.00009958065,0.0000026296498,0.00004325749,0.0000116030915,0.000009056096,0.000033832173,0.0004920941,0.98568743,0.01356907],"study_design_scores_gemma":[0.000009336757,0.000006742777,0.00008394535,0.00016607333,0.000003896667,0.00009386893,0.00003666846,0.000019030951,0.000049570284,0.00044545365,0.99908113,0.0000043056466],"about_ca_topic_score_codex":0.0013659898,"about_ca_topic_score_gemma":0.0020785024,"teacher_disagreement_score":0.5975977,"about_ca_system_score_codex":0.0022022275,"about_ca_system_score_gemma":0.005155496,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4390781999","doi":"10.1002/asi.24865","title":"<scp>Phenomenon‐based</scp> classification: An Annual Review of Information Science and Technology (ARIST) paper","year":2024,"lang":"en","type":"article","venue":"Journal of the Association for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Phenomenon; Identification (biology); Computer science; Discipline; Epistemology; Information system; Data science; Sociology; Social science; Engineering; Philosophy","score_opus":0.009216745243985884,"score_gpt":0.28738744698619406,"score_spread":0.2781707017422082,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390781999","genre_codex":"review","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00032220365,0.93751514,0.0029513305,0.040905092,0.0038981233,0.000018073304,0.00007817991,0.00003737316,0.014274492],"genre_scores_gemma":[0.012871949,0.9689507,0.003079869,0.0043178997,0.006572455,0.000050642797,0.00015290409,0.000047282538,0.003956429],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9945602,0.0018865236,0.0005304489,0.00049869006,0.0023308357,0.0001933182],"domain_scores_gemma":[0.9804081,0.010036988,0.0013761187,0.00093413837,0.0066698454,0.00057488994],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0091981385,0.0007079975,0.0011639871,0.02322411,0.0020391257,0.009872334,0.0011835492,0.003276092,0.00448628],"category_scores_gemma":[0.01712643,0.0005449101,0.0007252773,0.031042613,0.010278159,0.014786643,0.0024941757,0.004787075,0.0019423981],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015705706,0.000031253952,0.0010623451,0.0030647477,0.000069537135,0.000056981247,0.0009394988,0.0006523479,0.00015183527,0.41471952,0.2873404,0.29189584],"study_design_scores_gemma":[0.00000409229,0.00001804276,0.0029153775,0.0075275023,0.0000360288,0.00015425704,0.0006805553,0.00045487288,0.00008206895,0.08452306,0.9035727,0.000031516836],"about_ca_topic_score_codex":0.009100491,"about_ca_topic_score_gemma":0.007355443,"teacher_disagreement_score":0.02322411,"about_ca_system_score_codex":0.008713773,"about_ca_system_score_gemma":0.0090874415,"threshold_uncertainty_score":0.06322312},"labels":[],"label_agreement":null},{"id":"W4391621552","doi":"10.2139/ssrn.4719403","title":"Comprehensive Analysis of Transformer Networks in Identifying Informative Sentences Containing Customer Needs","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Transformer; Computer science; Customer needs; Natural language processing; Business; Engineering; Marketing; Electrical engineering","score_opus":0.0169723683184902,"score_gpt":0.30422782540503407,"score_spread":0.2872554570865439,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391621552","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.56640506,0.0034111568,0.40768367,0.0011076452,0.000104888204,0.00037294702,0.007571613,0.0024169146,0.010926059],"genre_scores_gemma":[0.91833264,0.00092576985,0.07016623,0.0000883193,0.00010226309,0.00014108123,0.006992274,0.00016732796,0.0030839974],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990728,0.0004009686,0.000060682345,0.00018485647,0.00020171046,0.000078874415],"domain_scores_gemma":[0.9916038,0.006623833,0.00035652937,0.00036794966,0.0008800714,0.00016786432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019696483,0.0007629989,0.00053368445,0.0037300647,0.0007528924,0.0012263007,0.00058118603,0.0008548969,0.0028820042],"category_scores_gemma":[0.008697625,0.0003083328,0.0005084104,0.0025001087,0.0003292905,0.0028767732,0.0007936505,0.00092706375,0.0011918077],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019825434,0.0007531477,0.048346713,0.0013400303,0.0005257507,0.0019875772,0.0019798304,0.0794128,0.08592621,0.027241677,0.020738393,0.72976536],"study_design_scores_gemma":[0.000047926154,0.0002625469,0.023636987,0.00008676228,0.00046366733,0.00091605546,0.00084720715,0.91681993,0.020273926,0.028635794,0.007958315,0.00005097588],"about_ca_topic_score_codex":0.002660104,"about_ca_topic_score_gemma":0.006552298,"teacher_disagreement_score":0.0037300647,"about_ca_system_score_codex":0.0005020112,"about_ca_system_score_gemma":0.0010465876,"threshold_uncertainty_score":0.010416627},"labels":[],"label_agreement":null},{"id":"W4391756409","doi":"10.1016/j.mehy.2024.111295","title":"Comment on: Hypothesis testing of the adoption of pseudoscientific methods","year":2024,"lang":"en","type":"article","venue":"Medical Hypotheses","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Children's Hospital of Eastern Ontario; Royal Ottawa Mental Health Centre; University of Ottawa","funders":"","keywords":"Pseudoscience; Psychology; Medicine; Alternative medicine; Pathology","score_opus":0.07637725113718358,"score_gpt":0.35444273570711243,"score_spread":0.27806548456992886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391756409","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0003616648,0.00040456647,0.00048505949,0.97401094,0.023250967,0.000053425872,0.00017288438,0.000056962035,0.0012034804],"genre_scores_gemma":[0.001890437,0.00014671468,0.0005824303,0.9823214,0.013782305,0.00010522458,0.000023893614,0.00002319387,0.0011244692],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.96234,0.012616274,0.006490515,0.0060138986,0.010056527,0.0024827863],"domain_scores_gemma":[0.6585027,0.268565,0.015509595,0.008120944,0.041970532,0.00733129],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.042490404,0.0017175351,0.0028676316,0.0031766258,0.0074706995,0.0060553057,0.010945888,0.09887192,0.015545676],"category_scores_gemma":[0.3100032,0.0017507059,0.0046001673,0.0028335496,0.017699387,0.008623024,0.005668089,0.07820505,0.012277085],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022745569,0.000038766735,0.0011835657,0.00027759274,0.00012127476,0.0008632483,0.0012079339,0.00014277031,0.00045210772,0.01285633,0.978819,0.0038100597],"study_design_scores_gemma":[0.0005437754,0.0001543784,0.004714305,0.0017004653,0.0003686368,0.0013408784,0.0034145426,0.0014662605,0.0018830713,0.03685727,0.9472553,0.00030115564],"about_ca_topic_score_codex":0.025809778,"about_ca_topic_score_gemma":0.021886142,"teacher_disagreement_score":0.9575096,"about_ca_system_score_codex":0.0095291985,"about_ca_system_score_gemma":0.0106702605,"threshold_uncertainty_score":0.22471344},"labels":[],"label_agreement":null},{"id":"W4391793582","doi":"10.1145/3632971.3632980","title":"Open-ended questions automated evaluation: proposal of a new generation","year":2023,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science","score_opus":0.08351453324136883,"score_gpt":0.40420583955243855,"score_spread":0.32069130631106973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391793582","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0869879,0.0011655862,0.862425,0.0043370244,0.001971124,0.010058557,0.0014859572,0.00977538,0.02179338],"genre_scores_gemma":[0.21930996,0.0004793994,0.735542,0.0008755727,0.0006296456,0.0061933664,0.004966243,0.0011078508,0.030895943],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.980724,0.009325071,0.0014164614,0.0020568874,0.0058815912,0.0005960136],"domain_scores_gemma":[0.9697119,0.007063117,0.0006458586,0.0037006726,0.017367514,0.0015109446],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02681087,0.0012435829,0.0010958993,0.0035856327,0.0010459856,0.0036960563,0.0029217866,0.0024410866,0.018185945],"category_scores_gemma":[0.031853195,0.0005622427,0.0012157874,0.0010863537,0.0013852431,0.0033123645,0.002974218,0.0013424295,0.0051629213],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014287292,0.0016382523,0.009252296,0.00069476943,0.00014068115,0.0006130944,0.0020686474,0.004326335,0.020170344,0.011125672,0.03907118,0.9094701],"study_design_scores_gemma":[0.0026912123,0.010956778,0.07137405,0.0020409157,0.0012008583,0.0062065525,0.0031018294,0.33544123,0.10472667,0.032914978,0.4285259,0.0008189509],"about_ca_topic_score_codex":0.0027675484,"about_ca_topic_score_gemma":0.0015983857,"teacher_disagreement_score":0.02681087,"about_ca_system_score_codex":0.0015162353,"about_ca_system_score_gemma":0.00247832,"threshold_uncertainty_score":0.1417911},"labels":[],"label_agreement":null},{"id":"W4391971213","doi":"10.32942/x2vg87","title":"The changing landscape of text mining - a review of approaches for ecology and evolution","year":2024,"lang":"en","type":"review","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Medical Research Council","keywords":"Ecology; Geography; Landscape ecology; Environmental resource management; Biology; Environmental science","score_opus":0.06786742054177088,"score_gpt":0.3466628727282187,"score_spread":0.27879545218644786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391971213","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0001168556,0.9931908,0.0023866952,0.0030896813,0.00035034036,0.000018405095,0.000060628743,0.000034847584,0.0007517867],"genre_scores_gemma":[0.0008377349,0.993356,0.0038987356,0.0011173304,0.0004337449,0.00003868985,0.00007658809,0.000016983588,0.00022424036],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.996482,0.0012944367,0.0006072133,0.00044061514,0.0010823566,0.00009322842],"domain_scores_gemma":[0.9625919,0.03169926,0.001467866,0.0006856657,0.0030528544,0.00050253415],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.008682546,0.0013704286,0.0026785044,0.013025141,0.0007816405,0.004099571,0.0025218537,0.002352465,0.0035042893],"category_scores_gemma":[0.019162608,0.0006597863,0.0016374135,0.014527733,0.003165174,0.009813263,0.0019576661,0.0033569152,0.0020846457],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030588082,0.000032308282,0.00031345277,0.037362386,0.00018265352,0.00007658533,0.00027955754,0.00045386344,0.00049353193,0.016411358,0.022133244,0.92223054],"study_design_scores_gemma":[0.000021499705,0.000098414755,0.0020961317,0.04565545,0.00027815616,0.0006969008,0.0005417341,0.00066284474,0.0006072944,0.04263101,0.90660053,0.00010991783],"about_ca_topic_score_codex":0.002806202,"about_ca_topic_score_gemma":0.0042240755,"teacher_disagreement_score":0.99131745,"about_ca_system_score_codex":0.0023912305,"about_ca_system_score_gemma":0.00537934,"threshold_uncertainty_score":0.045918286},"labels":[],"label_agreement":null},{"id":"W4392384312","doi":"10.1145/3616855.3635691","title":"Vector Search with OpenAI Embeddings: Lucene Is All You Need","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Universitas Brawijaya","keywords":"Computer science; Ranking (information retrieval); Encoder; Information retrieval; Architecture; Artificial intelligence; Data mining; Geography; Operating system","score_opus":0.018889090855615288,"score_gpt":0.3146775010715027,"score_spread":0.2957884102158874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392384312","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06391831,0.0038032418,0.3669608,0.004106849,0.0018148404,0.00077065383,0.03525756,0.4499768,0.07339091],"genre_scores_gemma":[0.29994825,0.0015631333,0.52164716,0.0025209538,0.00031932045,0.0012455768,0.0892548,0.0303852,0.053115703],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9976367,0.00054047874,0.00014837603,0.00044112004,0.0009891038,0.00024413779],"domain_scores_gemma":[0.99614584,0.0013326578,0.00011619896,0.0014946011,0.0007155306,0.00019499009],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001832191,0.0017820667,0.0011957831,0.0014476101,0.0009846977,0.0033880745,0.002780543,0.0019100097,0.039999202],"category_scores_gemma":[0.0134824505,0.000726535,0.000890805,0.0019200714,0.0011054187,0.0097787585,0.0035523758,0.0027920178,0.031514503],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016384685,0.00076870003,0.0031225928,0.0014198518,0.00027594666,0.0008080768,0.0006791344,0.022600988,0.021296194,0.028848192,0.4908368,0.42770502],"study_design_scores_gemma":[0.0006749767,0.001079033,0.0032517854,0.00041093113,0.00009877181,0.0008313289,0.00075483765,0.4781065,0.070692144,0.06556775,0.37818357,0.00034843388],"about_ca_topic_score_codex":0.009623748,"about_ca_topic_score_gemma":0.023257844,"teacher_disagreement_score":0.039999202,"about_ca_system_score_codex":0.00095592544,"about_ca_system_score_gemma":0.0011115157,"threshold_uncertainty_score":0.1338107},"labels":[],"label_agreement":null},{"id":"W4392612056","doi":"10.1145/3627508.3638307","title":"Visual Keyword/Result Linking: Using Interaction to Dynamically Reveal Relationships","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Computer science; Information retrieval; Search engine; Space (punctuation); Visualization; Digital library; Interface (matter); User satisfaction; Value (mathematics); Information visualization; Feature (linguistics); World Wide Web; Human–computer interaction; Data mining; Machine learning","score_opus":0.04515645247046382,"score_gpt":0.3678306697881741,"score_spread":0.3226742173177103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392612056","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14974694,0.0024703122,0.81292415,0.00072145194,0.00019393944,0.0008121782,0.0010637068,0.020208517,0.01185886],"genre_scores_gemma":[0.37984422,0.0011888864,0.6110198,0.00036049663,0.00014153,0.0006977674,0.00096364756,0.0013995285,0.0043840404],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99710816,0.0014776556,0.00026150164,0.00048711925,0.00054769835,0.00011786825],"domain_scores_gemma":[0.98039985,0.014787522,0.0014394659,0.0018955304,0.0010158978,0.0004617627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032281056,0.0017296894,0.0007876611,0.0034678963,0.0004562376,0.003776913,0.0013276739,0.0012810554,0.0058307657],"category_scores_gemma":[0.021755604,0.00048141493,0.00080906483,0.0023359193,0.00095220946,0.006601657,0.0033059665,0.0007527457,0.002246263],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028800373,0.0005809792,0.009570926,0.0037462448,0.00025214467,0.00073217996,0.016957102,0.004304283,0.13553397,0.013550099,0.008761129,0.80313087],"study_design_scores_gemma":[0.0015069718,0.009407821,0.052806124,0.0040455675,0.0022774762,0.006946645,0.016982738,0.17218779,0.2761458,0.14761665,0.30838642,0.001689942],"about_ca_topic_score_codex":0.00051947415,"about_ca_topic_score_gemma":0.0006462319,"teacher_disagreement_score":0.0058307657,"about_ca_system_score_codex":0.00037546435,"about_ca_system_score_gemma":0.0005482874,"threshold_uncertainty_score":0.019505858},"labels":[],"label_agreement":null},{"id":"W4392811535","doi":"10.1111/cogs.13416","title":"Probing the Representational Structure of Regular Polysemy via Sense Analogy Questions: Insights from Contextual Word Vectors","year":2024,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto","keywords":"Polysemy; Analogy; Computer science; Natural language processing; Linguistics; Artificial intelligence; Mathematics; Cognitive psychology; Psychology","score_opus":0.014178112154821837,"score_gpt":0.30500564028455596,"score_spread":0.2908275281297341,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392811535","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7053079,0.00051779056,0.28477615,0.00082322943,0.000066766115,0.00008954885,0.0013613249,0.0007094056,0.0063478095],"genre_scores_gemma":[0.9738045,0.00010220068,0.02466332,0.000060372462,0.000016971117,0.000044986507,0.0006927095,0.00011145317,0.0005034915],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99859935,0.0006809949,0.00009006659,0.00039152743,0.00014654832,0.000091527785],"domain_scores_gemma":[0.99205023,0.0054617445,0.00079312624,0.00084874046,0.00060498365,0.00024108506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022862658,0.0006836245,0.0003818501,0.0011106842,0.00048810343,0.0015074413,0.0005857786,0.0008130966,0.0047688833],"category_scores_gemma":[0.018637922,0.000369876,0.0007028087,0.0010608085,0.0013949549,0.0061571724,0.0019007945,0.0014380806,0.00065966113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020373974,0.00043817196,0.15780723,0.0018269183,0.00041646784,0.0012216017,0.02615419,0.04546589,0.11685083,0.20648576,0.005732684,0.4355629],"study_design_scores_gemma":[0.00006974293,0.00036633402,0.06882182,0.0002469326,0.0001629257,0.0010079505,0.007927379,0.5453146,0.017557662,0.34615082,0.012224911,0.00014892843],"about_ca_topic_score_codex":0.0022518258,"about_ca_topic_score_gemma":0.0030884158,"teacher_disagreement_score":0.0047688833,"about_ca_system_score_codex":0.00067266077,"about_ca_system_score_gemma":0.00049497274,"threshold_uncertainty_score":0.015953481},"labels":[],"label_agreement":null},{"id":"W4392910034","doi":"10.1109/icassp48485.2024.10447992","title":"VIC-KD: Variance-Invariance-Covariance Knowledge Distillation to Make Keyword Spotting More Robust Against Adversarial Attacks","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Adversarial system; Covariance; Distillation; Variance (accounting); Computer science; Spotting; Artificial intelligence; Mathematics; Statistics; Chemistry; Chromatography","score_opus":0.02042862100012554,"score_gpt":0.30573294457778605,"score_spread":0.2853043235776605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392910034","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010313287,0.0005219526,0.98238075,0.00033279494,0.00012584914,0.00006283435,0.0003153626,0.004418652,0.0015285197],"genre_scores_gemma":[0.5078733,0.0005713484,0.47439685,0.0012395697,0.00023779861,0.00024849598,0.0030686024,0.0011533562,0.01121061],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986035,0.000352376,0.000085265805,0.00040611834,0.0004265875,0.00012623626],"domain_scores_gemma":[0.99807173,0.00086867536,0.00015417727,0.00056801084,0.0002452173,0.00009218636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014944053,0.0013416173,0.001293033,0.00089634984,0.0005602754,0.0010053704,0.0021696244,0.0014309393,0.0033726303],"category_scores_gemma":[0.0070064124,0.00045422805,0.0009447721,0.0008571967,0.0015634514,0.0028180291,0.0031160554,0.0028396107,0.0017523266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042938514,0.00023018054,0.0009308018,0.00027309664,0.00014552804,0.00022389407,0.00020464696,0.4532078,0.01986068,0.039371733,0.018157672,0.4669646],"study_design_scores_gemma":[0.000018524199,0.00005145022,0.00008562987,0.000010889918,0.000008477002,0.00006322916,0.000016107524,0.97653913,0.006899593,0.013914194,0.0023740833,0.000018647315],"about_ca_topic_score_codex":0.00318327,"about_ca_topic_score_gemma":0.00529229,"teacher_disagreement_score":0.0033726303,"about_ca_system_score_codex":0.0008437181,"about_ca_system_score_gemma":0.0016568485,"threshold_uncertainty_score":0.011282623},"labels":[],"label_agreement":null},{"id":"W4393012052","doi":"10.3389/frai.2024.1200949","title":"Key point generation as an instrument for generating core statements of a political debate on Twitter","year":2024,"lang":"en","type":"article","venue":"Frontiers in Artificial Intelligence","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Automatic summarization; Computer science; Key (lock); Process (computing); Artificial intelligence; Point (geometry); Ranking (information retrieval); Complement (music); Hyperparameter; Natural language processing; Information retrieval; GRASP; Probabilistic logic; Word embedding; Word (group theory); Data mining; Machine learning; Embedding","score_opus":0.1039080750419846,"score_gpt":0.3956751987625539,"score_spread":0.29176712372056934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393012052","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12403535,0.0009603669,0.8387242,0.0013367189,0.00034127428,0.0020909843,0.010959157,0.011768183,0.009783773],"genre_scores_gemma":[0.35673723,0.00038500832,0.6223173,0.0001323687,0.00021708962,0.0016862763,0.014156618,0.0006198177,0.0037483368],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9961599,0.001921675,0.00027004568,0.0005770044,0.0009587995,0.000112509864],"domain_scores_gemma":[0.98567766,0.008892566,0.0012297768,0.0012802797,0.0026597385,0.00025991193],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038308147,0.0014404496,0.0006142667,0.005680986,0.0008602672,0.002066178,0.00089248334,0.0009078826,0.0044663018],"category_scores_gemma":[0.023435751,0.0003137941,0.0007621087,0.003591986,0.00061019004,0.002893587,0.0021403336,0.0010766377,0.0028800825],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012229626,0.00028646764,0.011538945,0.002188686,0.00019055672,0.00054052245,0.0070304903,0.040106177,0.040215146,0.021206874,0.026271887,0.84920126],"study_design_scores_gemma":[0.0002134107,0.0007710124,0.015401024,0.00042779028,0.00019448064,0.0004085368,0.0077417362,0.76100194,0.07069805,0.061919153,0.08099266,0.00023015794],"about_ca_topic_score_codex":0.0010358997,"about_ca_topic_score_gemma":0.0022003078,"teacher_disagreement_score":0.005680986,"about_ca_system_score_codex":0.00085158873,"about_ca_system_score_gemma":0.0009981247,"threshold_uncertainty_score":0.02025956},"labels":[],"label_agreement":null},{"id":"W4393347796","doi":"10.1007/s41019-023-00239-2","title":"Uncovering Flat and Hierarchical Topics by Community Discovery on Word Co-occurrence Network","year":2024,"lang":"en","type":"article","venue":"Data Science and Engineering","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke; University of Alberta","funders":"Centre National de la Recherche Scientifique; Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Hierarchy; Identification (biology); Topic model; Word (group theory); Exploit; Data science; Tree (set theory); Information retrieval; Linguistics","score_opus":0.023266520834801834,"score_gpt":0.2920903359651108,"score_spread":0.26882381513030895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393347796","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10570736,0.0025527207,0.88523805,0.00045308008,0.00007654077,0.0003280116,0.0016259494,0.001611414,0.0024068758],"genre_scores_gemma":[0.5231264,0.0014504587,0.4648589,0.00012546014,0.00024592894,0.00043918742,0.0065794517,0.00027951936,0.0028946784],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981628,0.00049709855,0.000100170044,0.00061034964,0.0004274637,0.00020216775],"domain_scores_gemma":[0.9953041,0.002745843,0.00067224674,0.00036066226,0.0006919269,0.00022513827],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001791662,0.0013981748,0.0012200157,0.011115947,0.0015237153,0.0016083416,0.0013337167,0.0012995952,0.00083327875],"category_scores_gemma":[0.0068435282,0.000492062,0.0014540373,0.006835887,0.0008192831,0.0039798818,0.0021311024,0.0012315759,0.00076365203],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010034158,0.00056594465,0.052075822,0.0011625905,0.00065513887,0.0012323642,0.0034875884,0.11048073,0.025864284,0.028577656,0.01928394,0.7556105],"study_design_scores_gemma":[0.000041220825,0.00006823376,0.005045797,0.000075394186,0.0001326399,0.00044302494,0.0006327314,0.9463732,0.0050706523,0.03491358,0.0071538663,0.00004963932],"about_ca_topic_score_codex":0.006110988,"about_ca_topic_score_gemma":0.008862048,"teacher_disagreement_score":0.011115947,"about_ca_system_score_codex":0.00077474845,"about_ca_system_score_gemma":0.0012281577,"threshold_uncertainty_score":0.012150824},"labels":[],"label_agreement":null},{"id":"W4393676817","doi":"10.5281/zenodo.10069275","title":"The Effect of Typing Efficiency and Suggestion Accuracy on Usage of Word Suggestions and Entry Speed","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Word (group theory); Typing; Computer science; Natural language processing; Arithmetic; Psychology; Speech recognition; Linguistics; Mathematics; Philosophy","score_opus":0.01859516809902201,"score_gpt":0.280757001738307,"score_spread":0.26216183363928497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393676817","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13933371,0.0007229978,0.0019795485,0.000427324,0.00023750786,0.00049090805,0.84825325,0.0053162747,0.0032384424],"genre_scores_gemma":[0.041695785,0.00016846739,0.005734242,0.00013277962,0.000021853039,0.0008735278,0.9486167,0.00021886523,0.002537798],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9976031,0.0005121084,0.0005065989,0.00054697896,0.00062103965,0.00021017305],"domain_scores_gemma":[0.9897181,0.004983477,0.000753926,0.0017382967,0.0023434209,0.00046280143],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020315186,0.0017353625,0.0010946227,0.0019844642,0.00070963457,0.001250248,0.0017138827,0.0013562487,0.0049047726],"category_scores_gemma":[0.012230389,0.00038417045,0.00084234163,0.0022630126,0.00044547886,0.0012708756,0.0012230288,0.001356487,0.009075632],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007914999,0.0039664973,0.049793914,0.006771856,0.0005609824,0.00034642563,0.0014614596,0.0059521417,0.017941348,0.00086182007,0.792052,0.11237669],"study_design_scores_gemma":[0.004440413,0.0030375053,0.33815506,0.0012020908,0.0011972913,0.0017358005,0.0026801012,0.062282085,0.05078186,0.0029113565,0.5308995,0.00067687803],"about_ca_topic_score_codex":0.011183103,"about_ca_topic_score_gemma":0.0247881,"teacher_disagreement_score":0.011183103,"about_ca_system_score_codex":0.0008556381,"about_ca_system_score_gemma":0.0014575092,"threshold_uncertainty_score":0.02223605},"labels":[],"label_agreement":null},{"id":"W4393719360","doi":"10.5281/zenodo.10069276","title":"The Effect of Typing Efficiency and Suggestion Accuracy on Usage of Word Suggestions and Entry Speed","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Word (group theory); Computer science; Natural language processing; Typing; Speech recognition; Artificial intelligence; Linguistics; Philosophy","score_opus":0.01859516809902201,"score_gpt":0.280757001738307,"score_spread":0.26216183363928497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393719360","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13933371,0.0007229978,0.0019795485,0.000427324,0.00023750786,0.00049090805,0.84825325,0.0053162747,0.0032384424],"genre_scores_gemma":[0.041695785,0.00016846739,0.005734242,0.00013277962,0.000021853039,0.0008735278,0.9486167,0.00021886523,0.002537798],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9976031,0.0005121084,0.0005065989,0.00054697896,0.00062103965,0.00021017305],"domain_scores_gemma":[0.9897181,0.004983477,0.000753926,0.0017382967,0.0023434209,0.00046280143],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020315186,0.0017353625,0.0010946227,0.0019844642,0.00070963457,0.001250248,0.0017138827,0.0013562487,0.0049047726],"category_scores_gemma":[0.012230389,0.00038417045,0.00084234163,0.0022630126,0.00044547886,0.0012708756,0.0012230288,0.001356487,0.009075632],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007914999,0.0039664973,0.049793914,0.006771856,0.0005609824,0.00034642563,0.0014614596,0.0059521417,0.017941348,0.00086182007,0.792052,0.11237669],"study_design_scores_gemma":[0.004440413,0.0030375053,0.33815506,0.0012020908,0.0011972913,0.0017358005,0.0026801012,0.062282085,0.05078186,0.0029113565,0.5308995,0.00067687803],"about_ca_topic_score_codex":0.011183103,"about_ca_topic_score_gemma":0.0247881,"teacher_disagreement_score":0.011183103,"about_ca_system_score_codex":0.0008556381,"about_ca_system_score_gemma":0.0014575092,"threshold_uncertainty_score":0.02223605},"labels":[],"label_agreement":null},{"id":"W4393748815","doi":"10.5281/zenodo.5720363","title":"Replication Package for the Paper: Transfer Learning with Time Series Data: A Systematic Mapping Study","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Replication (statistics); Computer science; Series (stratigraphy); R package; Time series; Machine learning; Programming language; Biology; Statistics; Mathematics","score_opus":0.042432545873847614,"score_gpt":0.27273189811639115,"score_spread":0.23029935224254353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393748815","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006776082,0.0012651124,0.09950057,0.005198475,0.0094292145,0.13242854,0.7224475,0.012641712,0.010312824],"genre_scores_gemma":[0.014602306,0.0003497072,0.1732596,0.001647626,0.00036942767,0.7334896,0.06650694,0.003266591,0.0065082083],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.92938566,0.041717168,0.012891924,0.0069652647,0.0072247516,0.0018153334],"domain_scores_gemma":[0.68230754,0.19291922,0.0125809405,0.077272736,0.033251043,0.0016685643],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.09664204,0.0029758008,0.0037749272,0.0076630386,0.003045875,0.004885345,0.0040532597,0.0035462978,0.24624784],"category_scores_gemma":[0.46944577,0.002674533,0.01060477,0.011154744,0.0023225935,0.0044303485,0.0056458665,0.0048711463,0.041752264],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037307225,0.00045153694,0.0032608297,0.05545082,0.0049378793,0.00032394705,0.0024380065,0.001770045,0.0010981569,0.014004101,0.85073864,0.06179535],"study_design_scores_gemma":[0.022547377,0.001548427,0.019916305,0.023916217,0.009421926,0.0004572358,0.0021639091,0.005026899,0.0031044804,0.053349968,0.85751355,0.0010337512],"about_ca_topic_score_codex":0.003796078,"about_ca_topic_score_gemma":0.006981803,"teacher_disagreement_score":0.24624784,"about_ca_system_score_codex":0.0026602636,"about_ca_system_score_gemma":0.013447543,"threshold_uncertainty_score":0.82378113},"labels":[],"label_agreement":null},{"id":"W4393753614","doi":"10.5281/zenodo.3635095","title":"Testing ritual knot tracing for cognitive priming effects rules out analytic analogy: Core Data Sets","year":2019,"lang":"en","type":"dataset","venue":"Figshare","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Analogy; Knot (papermaking); Tracing; Priming (agriculture); Core (optical fiber); Cognition; Computer science; Cognitive science; Psychology; Epistemology; Philosophy; Programming language; Engineering; Neuroscience; Biology","score_opus":0.23788761834435018,"score_gpt":0.4032536182026153,"score_spread":0.1653659998582651,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393753614","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028522552,0.00009381018,0.0016945725,0.0004947298,0.00010335145,0.00085074076,0.98823905,0.00052797305,0.0051435693],"genre_scores_gemma":[0.011594196,0.00006400621,0.007780261,0.00040424938,0.00004075927,0.01206669,0.96408457,0.000711596,0.0032535554],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9923252,0.002339058,0.0015266194,0.0017400464,0.0015270263,0.00054201717],"domain_scores_gemma":[0.94963896,0.022472227,0.003536571,0.014469296,0.008917139,0.00096584944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009570471,0.0013072933,0.0010543986,0.0040325946,0.0021550776,0.0033985563,0.003398826,0.0023227513,0.07035175],"category_scores_gemma":[0.074073434,0.0007975803,0.001381958,0.004145145,0.0011955746,0.002255922,0.0038559807,0.0026496663,0.043721493],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003205056,0.00012751023,0.0050912895,0.0017778924,0.000094944706,0.00005178725,0.0005390435,0.0002527075,0.00049032003,0.004529105,0.9784534,0.008271598],"study_design_scores_gemma":[0.0010449035,0.00007626727,0.024597822,0.0013347368,0.00012567276,0.00010645951,0.0008240056,0.0006611335,0.0015631212,0.007857139,0.96170294,0.00010576874],"about_ca_topic_score_codex":0.008976626,"about_ca_topic_score_gemma":0.02348938,"teacher_disagreement_score":0.07035175,"about_ca_system_score_codex":0.001809863,"about_ca_system_score_gemma":0.0044073607,"threshold_uncertainty_score":0.23535007},"labels":[],"label_agreement":null},{"id":"W4393792028","doi":"10.5281/zenodo.3635094","title":"Testing ritual knot tracing for cognitive priming effects rules out analytic analogy: Core Data Sets","year":2019,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Analogy; Knot (papermaking); Priming (agriculture); Cognition; Tracing; Core (optical fiber); Cognitive psychology; Computer science; Psychology; Epistemology; Neuroscience; Philosophy; Engineering; Programming language; Biology","score_opus":0.1521145632011919,"score_gpt":0.3529259361354704,"score_spread":0.2008113729342785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393792028","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011295811,0.00020686939,0.0041973894,0.000553806,0.0001527907,0.001722817,0.9726151,0.0007236996,0.008531694],"genre_scores_gemma":[0.020796347,0.00007341344,0.009406491,0.00035300606,0.000043025124,0.011717865,0.9536608,0.0005680677,0.0033809221],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9917762,0.002532735,0.0015168386,0.0018742544,0.0016688408,0.0006311466],"domain_scores_gemma":[0.9625677,0.014501211,0.002828643,0.012875965,0.0063655637,0.000860942],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009388265,0.001281408,0.0009859735,0.0035587652,0.0020045624,0.003116346,0.0028138729,0.0021504804,0.033857916],"category_scores_gemma":[0.052808248,0.0006929513,0.0012209313,0.0031857723,0.0013381846,0.0018075248,0.0036462257,0.0023981838,0.029760746],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008795705,0.0004016186,0.012116526,0.0025975332,0.00017914556,0.00013479353,0.001117976,0.00060533686,0.0018915326,0.009051164,0.95219386,0.018830925],"study_design_scores_gemma":[0.001234422,0.00015159133,0.042165525,0.0011016357,0.0001503061,0.0002335046,0.001175656,0.0012502124,0.0044830297,0.009255773,0.9386714,0.00012699905],"about_ca_topic_score_codex":0.0063273986,"about_ca_topic_score_gemma":0.014110078,"teacher_disagreement_score":0.033857916,"about_ca_system_score_codex":0.0016387337,"about_ca_system_score_gemma":0.0036448557,"threshold_uncertainty_score":0.11326599},"labels":[],"label_agreement":null},{"id":"W4393981161","doi":"10.25144/16085","title":"THE CONTRIBUTION OF AUTOMATIC SPEECH RECOGNITION FOR KEYWORDS TO ASSIST IN THE INTEGRATED ORGANISATION OF DIGITAL MESSAGES","year":2023,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kwantlen Polytechnic University","funders":"","keywords":"Computer science; Speech recognition; Natural language processing","score_opus":0.022005704841387256,"score_gpt":0.3041059391850769,"score_spread":0.28210023434368964,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393981161","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.215043,0.011285728,0.7389619,0.0026461233,0.002361875,0.0003082624,0.002041445,0.008286199,0.019065501],"genre_scores_gemma":[0.5546114,0.0034487902,0.41901886,0.0007404689,0.00089476863,0.00010518932,0.0023069896,0.00058511103,0.018288424],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991308,0.00028471375,0.000065916305,0.00020764851,0.00023743305,0.00007361726],"domain_scores_gemma":[0.9965403,0.0018938,0.00012214627,0.0002900848,0.00105798,0.000095718955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011151646,0.0008467808,0.00049828005,0.0015306708,0.00031735052,0.0016870213,0.0006448947,0.0009919646,0.004129036],"category_scores_gemma":[0.003789745,0.0002457429,0.0004172384,0.00064417196,0.00037898077,0.0013844406,0.000408479,0.0007278378,0.0046719606],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052418263,0.00015831158,0.0023986646,0.0003036033,0.00006447338,0.00015144682,0.00016803168,0.0027174444,0.22660014,0.0017586534,0.005003907,0.76015115],"study_design_scores_gemma":[0.000089894296,0.0009569175,0.015444985,0.00017796984,0.00036175305,0.0014219797,0.0005875371,0.5088978,0.41354042,0.0077913823,0.050552767,0.00017660102],"about_ca_topic_score_codex":0.0025549752,"about_ca_topic_score_gemma":0.0023692336,"teacher_disagreement_score":0.004129036,"about_ca_system_score_codex":0.00026341606,"about_ca_system_score_gemma":0.00061490637,"threshold_uncertainty_score":0.013812959},"labels":[],"label_agreement":null},{"id":"W4394842830","doi":"10.1515/cllt-2023-0028","title":"The distributional properties of long nominal compounds in scientific articles: an investigation based on the uniform information density hypothesis","year":2024,"lang":"en","type":"article","venue":"Corpus Linguistics and Linguistic Theory","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Constant (computer programming); Scientific literature; Information transmission; Econometrics; Data science; Information retrieval; Mathematics; Biology","score_opus":0.021486894088511916,"score_gpt":0.23154092932942288,"score_spread":0.21005403524091096,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394842830","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9656883,0.00090488466,0.020291746,0.0013120057,0.000051159415,0.000072268915,0.00043381413,0.00008207357,0.011163758],"genre_scores_gemma":[0.99647266,0.00013959428,0.0028284632,0.000044250977,0.00005091493,0.000039151324,0.00013343981,0.000025128109,0.00026636614],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9936094,0.0031281826,0.0007673131,0.0007982272,0.0014927362,0.0002040808],"domain_scores_gemma":[0.8100759,0.15244135,0.020273605,0.0055140518,0.010134092,0.001560998],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0079396125,0.00015461764,0.0004925621,0.0055032787,0.001959468,0.0037842188,0.0007074168,0.0010038716,0.0033614328],"category_scores_gemma":[0.0797412,0.00034747034,0.0002643646,0.005582053,0.0044397046,0.006436333,0.0019623628,0.0010866666,0.000447745],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034019863,0.0005078787,0.40524542,0.0025670188,0.00053180783,0.003916723,0.11360978,0.003915786,0.06887725,0.24953717,0.0047495672,0.14313965],"study_design_scores_gemma":[0.00019521077,0.00089178694,0.64428234,0.00067376334,0.00039117478,0.0044859573,0.08067863,0.041760683,0.024717968,0.17130892,0.030203287,0.0004103051],"about_ca_topic_score_codex":0.0011975225,"about_ca_topic_score_gemma":0.0011318636,"teacher_disagreement_score":0.9944967,"about_ca_system_score_codex":0.0013323944,"about_ca_system_score_gemma":0.00067019305,"threshold_uncertainty_score":0.041989148},"labels":[],"label_agreement":null},{"id":"W4394938781","doi":"10.5267/j.ijdns.2024.1.014","title":"Multi-objective of wind-driven optimization as feature selection and clustering to enhance text clustering","year":2024,"lang":"en","type":"article","venue":"International Journal of Data and Network Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Cluster analysis; Feature selection; Computer science; Selection (genetic algorithm); Artificial intelligence; Feature (linguistics); Pattern recognition (psychology); Correlation clustering; Data mining; Machine learning","score_opus":0.015719585447118743,"score_gpt":0.34774977221899417,"score_spread":0.3320301867718754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394938781","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10649468,0.00042596748,0.8894286,0.00020520485,0.00010145006,0.00016663397,0.00009303581,0.00038369646,0.0027007645],"genre_scores_gemma":[0.7541824,0.00019215458,0.24251923,0.00010164196,0.00004256126,0.0003043279,0.00021253676,0.000059238344,0.0023859239],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99961275,0.00010916424,0.00003285607,0.00007791883,0.00011856618,0.00004868564],"domain_scores_gemma":[0.9995492,0.00017276556,0.000063693544,0.000028198065,0.00016334868,0.000022693917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007958816,0.0008881895,0.0007984271,0.0010616594,0.00047200333,0.0006967696,0.00064707996,0.0007516096,0.00078045303],"category_scores_gemma":[0.0015127568,0.00030099176,0.0007365012,0.00089318596,0.00030314157,0.0006707525,0.000448077,0.0004338409,0.00015129735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009895504,0.00010945618,0.001654847,0.00014069176,0.000091728514,0.000102808684,0.00008520286,0.90443796,0.0093257,0.002452634,0.0011344386,0.08036548],"study_design_scores_gemma":[0.000008368264,0.00004243692,0.00032908333,0.0000042176384,0.000009121079,0.000012922636,0.00001285562,0.9975738,0.0012602607,0.00042779773,0.000314478,0.0000047262884],"about_ca_topic_score_codex":0.0036944689,"about_ca_topic_score_gemma":0.0035495006,"teacher_disagreement_score":0.0036944689,"about_ca_system_score_codex":0.00059769786,"about_ca_system_score_gemma":0.0008402006,"threshold_uncertainty_score":0.007345915},"labels":[],"label_agreement":null},{"id":"W4394973187","doi":"10.48550/arxiv.2404.11793","title":"Enhancing Argument Summarization: Prioritizing Exhaustiveness in Key Point Generation and Introducing an Automatic Coverage Evaluation Metric","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada","keywords":"Automatic summarization; Argument (complex analysis); Key (lock); Metric (unit); Point (geometry); Computer science; Process management; Business; Mathematics; Operations management; Information retrieval; Engineering; Computer security; Medicine","score_opus":0.054917207792881305,"score_gpt":0.24167628664611276,"score_spread":0.18675907885323145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394973187","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055228658,0.002687978,0.9198565,0.00079529564,0.00025208763,0.00096392026,0.0023377927,0.011767341,0.0061105182],"genre_scores_gemma":[0.2518607,0.00081642205,0.7358916,0.00021029053,0.00021944677,0.0006804851,0.006248045,0.00165547,0.0024174638],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9880062,0.004828663,0.0014183163,0.0015280494,0.0038824282,0.00033631388],"domain_scores_gemma":[0.93748516,0.039358687,0.005283743,0.0059663556,0.01099034,0.00091570045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009595109,0.0019913826,0.0013904481,0.009980437,0.0009815355,0.004734751,0.0017845494,0.0021790597,0.0037056361],"category_scores_gemma":[0.075074755,0.0005168486,0.0010201713,0.0045799143,0.001237256,0.006482321,0.0030704602,0.0016369668,0.0026412609],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079243095,0.0003572175,0.009095262,0.0023490689,0.00026560153,0.00034772107,0.0021666058,0.022049805,0.050933354,0.012915316,0.019144172,0.8795835],"study_design_scores_gemma":[0.000299509,0.0015094518,0.017385498,0.00080699136,0.0006605163,0.0014242289,0.0017751914,0.6905593,0.17999242,0.04824061,0.056988284,0.0003578978],"about_ca_topic_score_codex":0.0014257086,"about_ca_topic_score_gemma":0.0019643912,"teacher_disagreement_score":0.009980437,"about_ca_system_score_codex":0.0012084391,"about_ca_system_score_gemma":0.001735788,"threshold_uncertainty_score":0.050744414},"labels":[],"label_agreement":null},{"id":"W4396531244","doi":"10.22215/etd/2024-15938","title":"The Fusion of Multilingual Semantic Search and Large Language Models: A New Paradigm for Enhanced Topic Exploration and Contextual Search","year":2024,"lang":"en","type":"dissertation","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Semantic search; Linguistics; Semantic Web","score_opus":0.035890965213824906,"score_gpt":0.3618835984187028,"score_spread":0.3259926332048779,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396531244","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04421541,0.0019082137,0.9468844,0.0009905226,0.00013126992,0.00008212038,0.00044543092,0.0021734124,0.0031692188],"genre_scores_gemma":[0.5607996,0.0016064601,0.43149522,0.00043574738,0.00031754127,0.00014881753,0.0016067327,0.00037325552,0.003216613],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986953,0.0005680555,0.00009896388,0.0002743107,0.00029503007,0.00006835997],"domain_scores_gemma":[0.99796474,0.0011556933,0.00014787438,0.0003592859,0.0003021399,0.00007023399],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016900533,0.000841691,0.00126799,0.0018213703,0.00043439274,0.0022220057,0.0007598436,0.0008444336,0.001697177],"category_scores_gemma":[0.006394734,0.0003482636,0.0012665392,0.0021722794,0.00054490333,0.006089199,0.0021376202,0.0014259001,0.0009226763],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076859863,0.00079684495,0.004445598,0.0005824008,0.0004273577,0.00029621224,0.0010968349,0.113057636,0.027919648,0.04086887,0.007931902,0.80180806],"study_design_scores_gemma":[0.000026364858,0.0001456294,0.0008009763,0.000029179482,0.00007109713,0.000105321546,0.00021100548,0.9535558,0.004146446,0.036404878,0.0044627003,0.000040595856],"about_ca_topic_score_codex":0.0038061556,"about_ca_topic_score_gemma":0.005954255,"teacher_disagreement_score":0.0038061556,"about_ca_system_score_codex":0.00070884917,"about_ca_system_score_gemma":0.0009463144,"threshold_uncertainty_score":0.0089380145},"labels":[],"label_agreement":null},{"id":"W4396993769","doi":"10.2139/ssrn.4830848","title":"Optimal Text-Based Time-Series Indices","year":2024,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke; Center for Interuniversity Research and Analysis on Organizations; HEC Montréal","funders":"","keywords":"Series (stratigraphy); Time series; Computer science; Econometrics; Mathematics; Statistics; Geology","score_opus":0.004212325084687087,"score_gpt":0.24914385377778328,"score_spread":0.2449315286930962,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396993769","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06033848,0.0016693856,0.9276016,0.0006737888,0.0004139442,0.00018851702,0.0015875687,0.0014800004,0.006046803],"genre_scores_gemma":[0.43937632,0.0013802141,0.54785156,0.00024286001,0.00070071407,0.00038419146,0.003961478,0.00053072954,0.005572],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9982742,0.0005090632,0.0002478925,0.00038472065,0.000448655,0.00013545163],"domain_scores_gemma":[0.9949092,0.0027936692,0.0004185439,0.00052618084,0.0011417709,0.00021058127],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022987844,0.0011681085,0.0016132056,0.003184316,0.0004925733,0.0024956628,0.0011565826,0.0013909738,0.0063858484],"category_scores_gemma":[0.017289486,0.0003723293,0.0007958649,0.0031122407,0.00052421924,0.0039117825,0.0012854983,0.0011780098,0.0027895523],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018534156,0.00058217306,0.0036062864,0.0005183352,0.00016892729,0.00012234398,0.000113683345,0.12836857,0.021283986,0.031869516,0.01193107,0.79958177],"study_design_scores_gemma":[0.0001158458,0.0003392338,0.0023937938,0.00006526595,0.00010805238,0.0001305133,0.00006554995,0.9502791,0.00928542,0.034343634,0.0028357496,0.000037810514],"about_ca_topic_score_codex":0.0008626896,"about_ca_topic_score_gemma":0.0012135948,"teacher_disagreement_score":0.0063858484,"about_ca_system_score_codex":0.00064781704,"about_ca_system_score_gemma":0.0017006035,"threshold_uncertainty_score":0.021362841},"labels":[],"label_agreement":null},{"id":"W4398946215","doi":"10.7910/dvn/wndofl","title":"Replication data for Identifying science in the news","year":2022,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; Simon Fraser University","funders":"","keywords":"Replication (statistics); Computer science; Computational biology; Data science; Information retrieval; Biology; Virology","score_opus":0.08390283603802001,"score_gpt":0.36997543838186797,"score_spread":0.28607260234384796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398946215","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0063197813,0.00015133512,0.0014456979,0.00039631923,0.0002448569,0.0012053384,0.9834568,0.0006712763,0.006108757],"genre_scores_gemma":[0.0068435,0.000060703784,0.0044707116,0.00022006253,0.00005934714,0.0051887506,0.9802699,0.00020067499,0.002686299],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98347074,0.0050643124,0.0027583495,0.002507361,0.005322971,0.00087619247],"domain_scores_gemma":[0.9322274,0.021925468,0.004732076,0.020688392,0.018730678,0.0016960491],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.013954208,0.0011105747,0.0011166411,0.005704674,0.0023391386,0.00260245,0.0030742486,0.00195335,0.0360999],"category_scores_gemma":[0.0837463,0.0006542859,0.0013833295,0.009994095,0.0013094149,0.0016545762,0.003301407,0.0026459051,0.038279407],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050875096,0.00032676748,0.008068633,0.0013122858,0.00009027878,0.00010512036,0.00064672664,0.0006080538,0.0006697439,0.0031354842,0.9687376,0.01579049],"study_design_scores_gemma":[0.00081598654,0.00016497463,0.025121575,0.00048962206,0.00008235923,0.00020870945,0.0010823269,0.0011717106,0.002091446,0.0024196075,0.96623933,0.000112499496],"about_ca_topic_score_codex":0.01745218,"about_ca_topic_score_gemma":0.03112199,"teacher_disagreement_score":0.99739754,"about_ca_system_score_codex":0.0024841735,"about_ca_system_score_gemma":0.0046903808,"threshold_uncertainty_score":0.12076622},"labels":[],"label_agreement":null},{"id":"W4399665456","doi":"10.18438/eblip30521","title":"Machine-learning Recommender Systems Can Inform Collection Development Decisions","year":2024,"lang":"en","type":"article","venue":"Evidence Based Library and Information Practice","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Recommender system; Information retrieval; Collaborative filtering; Cosine similarity; World Wide Web; Naive Bayes classifier; Artificial intelligence; Support vector machine; Cluster analysis","score_opus":0.021566165454653587,"score_gpt":0.28118580625305634,"score_spread":0.2596196407984028,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399665456","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055406783,0.06753797,0.5703095,0.11758877,0.005252559,0.003731548,0.023553988,0.0076073133,0.14901148],"genre_scores_gemma":[0.27472326,0.030704087,0.6469419,0.008711044,0.0031344208,0.0017173651,0.01760139,0.0012996403,0.015166929],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9451729,0.03294729,0.005468388,0.0066366578,0.008940301,0.0008344999],"domain_scores_gemma":[0.77846646,0.14318696,0.013131028,0.02196724,0.04041819,0.0028301738],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.06326904,0.00220893,0.0028116917,0.012704082,0.0032555892,0.0140093155,0.003286291,0.0034647093,0.019582367],"category_scores_gemma":[0.2714585,0.0019235783,0.00197048,0.013196424,0.0014559063,0.016046695,0.0040405234,0.004011099,0.01581813],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016323425,0.00028189912,0.03478518,0.0027287393,0.0007116499,0.00018524527,0.0019049062,0.0076844194,0.0006438664,0.027004708,0.18654796,0.7373583],"study_design_scores_gemma":[0.00031747954,0.00046530575,0.035066314,0.0048822,0.0011403026,0.00035797627,0.004621019,0.07388985,0.003014218,0.16031103,0.71517336,0.000761042],"about_ca_topic_score_codex":0.01531493,"about_ca_topic_score_gemma":0.034311377,"teacher_disagreement_score":0.9859907,"about_ca_system_score_codex":0.0039646043,"about_ca_system_score_gemma":0.006673637,"threshold_uncertainty_score":0.33460265},"labels":[],"label_agreement":null},{"id":"W4399828186","doi":"10.1007/978-3-031-54071-4_7","title":"Comparison of Information Structures and Their Blackwell Ordering","year":2024,"lang":"en","type":"book-chapter","venue":"Systems & control","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science","score_opus":0.011326714725242151,"score_gpt":0.2618874376106657,"score_spread":0.2505607228854235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399828186","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038628533,0.006856932,0.7301562,0.0032241521,0.00096982677,0.00013386009,0.0023890536,0.0011775842,0.21646397],"genre_scores_gemma":[0.38190004,0.007145372,0.5124819,0.00060237123,0.0010333031,0.0002790875,0.0032678198,0.0013353062,0.09195464],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9974848,0.0009020968,0.0003225249,0.00029921698,0.0008217728,0.00016961378],"domain_scores_gemma":[0.9785039,0.013215471,0.0010916424,0.0027400383,0.0037560083,0.0006928933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029554528,0.0006459099,0.00094714336,0.006023639,0.0020952588,0.009035445,0.001236499,0.0009780434,0.021231102],"category_scores_gemma":[0.025659828,0.00065683905,0.00082428736,0.0107433535,0.0029084329,0.013237788,0.0016003296,0.0019332101,0.003514413],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000052992214,0.000013217072,0.00023148562,0.00008000333,0.00000857658,0.000030692536,0.0001824256,0.0007136865,0.0003273724,0.9524529,0.004258001,0.041648757],"study_design_scores_gemma":[0.000011268656,0.000029682926,0.0005054359,0.00006267584,0.000017282156,0.00011338711,0.00018329348,0.00518247,0.00065684906,0.97330856,0.019911291,0.00001769334],"about_ca_topic_score_codex":0.0028487649,"about_ca_topic_score_gemma":0.0031403361,"teacher_disagreement_score":0.021231102,"about_ca_system_score_codex":0.0022804977,"about_ca_system_score_gemma":0.0021210245,"threshold_uncertainty_score":0.07102513},"labels":[],"label_agreement":null},{"id":"W4399855255","doi":"10.18280/isi.290327","title":"LAMBDA: Lexicon and Aspect-Based Multimodal Data Analysis of Tweet","year":2024,"lang":"fr","type":"article","venue":"Ingénierie des systèmes d information","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Lexicon; Natural language processing; Computer science; Lambda; Artificial intelligence; Physics; Optics","score_opus":0.03530651952822245,"score_gpt":0.3006870633019621,"score_spread":0.26538054377373965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399855255","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046894405,0.0007847938,0.9080068,0.0005658621,0.0003101535,0.0010159891,0.012966505,0.01984044,0.009615079],"genre_scores_gemma":[0.3228281,0.0008230755,0.63560367,0.00031123983,0.0004106799,0.0013587704,0.028953422,0.0011756463,0.008535474],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99890566,0.00023656506,0.00014705872,0.00023779804,0.00039236492,0.00008059121],"domain_scores_gemma":[0.99860746,0.0004237642,0.00022428759,0.00016911428,0.0005203263,0.000055116834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009394349,0.0012508698,0.0007222456,0.0056285355,0.00064239465,0.0021357941,0.00067779503,0.00059970847,0.0032587317],"category_scores_gemma":[0.0038623763,0.00033542764,0.0012644026,0.0032587969,0.00036730056,0.0021445814,0.0015164975,0.00086393254,0.003736123],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004814863,0.0002687047,0.012703531,0.00083028036,0.0003241627,0.0006817638,0.0013307937,0.00509567,0.07972358,0.0077118636,0.034614332,0.8562338],"study_design_scores_gemma":[0.00010672829,0.00049514126,0.039923728,0.00021418222,0.00034366534,0.0013787551,0.0026151275,0.76077247,0.05111351,0.03214972,0.110625975,0.00026107955],"about_ca_topic_score_codex":0.003072634,"about_ca_topic_score_gemma":0.005110424,"teacher_disagreement_score":0.0056285355,"about_ca_system_score_codex":0.0005701765,"about_ca_system_score_gemma":0.00088303874,"threshold_uncertainty_score":0.01090157},"labels":[],"label_agreement":null},{"id":"W4399885315","doi":"10.21203/rs.3.rs-4587452/v1","title":"Topic Composition in AEA Journals","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Composition (language); Literature; Art","score_opus":0.10093493111736679,"score_gpt":0.4908727773061603,"score_spread":0.3899378461887935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399885315","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89363366,0.025037631,0.0022700406,0.003008227,0.00095051294,0.000265537,0.033166546,0.00035975996,0.041308038],"genre_scores_gemma":[0.94816977,0.013349929,0.0037941958,0.00070794823,0.0014455799,0.00040279742,0.021091254,0.00022215393,0.010816299],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99182105,0.001624487,0.0017676617,0.0011997898,0.0030191245,0.00056787534],"domain_scores_gemma":[0.90139276,0.03995089,0.026507176,0.0035689971,0.022831228,0.0057488717],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.006094601,0.00040384056,0.00086467207,0.04195208,0.0011494943,0.00716992,0.00072522205,0.00076045026,0.0113155255],"category_scores_gemma":[0.051178645,0.00023352652,0.0006604374,0.044851284,0.00058330904,0.003123368,0.0027712488,0.0005533058,0.0038024257],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009361108,0.0002106836,0.74725795,0.00465037,0.00069602585,0.0006881958,0.008009151,0.0006704898,0.010025679,0.0070691956,0.03706129,0.1827248],"study_design_scores_gemma":[0.00005772534,0.00015698755,0.9153398,0.00079525524,0.0001910796,0.0012074035,0.0069701783,0.00086896453,0.0016486653,0.0035788803,0.06911986,0.00006526583],"about_ca_topic_score_codex":0.0011493592,"about_ca_topic_score_gemma":0.0012510421,"teacher_disagreement_score":0.9939054,"about_ca_system_score_codex":0.00079388754,"about_ca_system_score_gemma":0.0012745758,"threshold_uncertainty_score":0.037854195},"labels":[],"label_agreement":null},{"id":"W4399900193","doi":"10.18280/ria.380311","title":"Summarizing Business News: Evaluating BART, T5, and PEGASUS for Effective Information Extraction","year":2024,"lang":"en","type":"article","venue":"Revue d intelligence artificielle","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Information extraction; Computer science; Information retrieval","score_opus":0.03529687329299615,"score_gpt":0.3489899099083913,"score_spread":0.31369303661539516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399900193","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85776937,0.0150268935,0.0922628,0.0020197094,0.00066259847,0.00085074914,0.006318427,0.012923994,0.012165459],"genre_scores_gemma":[0.881746,0.0023309311,0.09709832,0.00029251908,0.00017910499,0.00021261637,0.013913962,0.00020310859,0.004023371],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987948,0.0004211821,0.00017145871,0.00023524035,0.0002953745,0.00008185827],"domain_scores_gemma":[0.99526924,0.0031204543,0.00034560886,0.0003102228,0.0006852022,0.00026926803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037153612,0.0011237593,0.0007917703,0.0030370047,0.00053366687,0.0013703,0.00084440986,0.0013444633,0.0013806538],"category_scores_gemma":[0.012740868,0.0002074546,0.0006628285,0.0017221008,0.0004632377,0.0028770482,0.0007483631,0.00093607407,0.0007910885],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004363158,0.0009719369,0.023608396,0.002000267,0.0007885015,0.000409424,0.00080234674,0.104461804,0.008801313,0.0026548912,0.025418844,0.8257192],"study_design_scores_gemma":[0.00035620164,0.0049453615,0.020491311,0.00025121667,0.00070277567,0.0004285853,0.0016243152,0.92620236,0.026836708,0.0028334612,0.015223665,0.00010410413],"about_ca_topic_score_codex":0.0067644403,"about_ca_topic_score_gemma":0.008204325,"teacher_disagreement_score":0.0067644403,"about_ca_system_score_codex":0.001065058,"about_ca_system_score_gemma":0.0011481254,"threshold_uncertainty_score":0.01964897},"labels":[],"label_agreement":null},{"id":"W4401043913","doi":"10.18653/v1/2024.naacl-long.454","title":"Enhancing Argument Summarization: Prioritizing Exhaustiveness in Key Point Generation and Introducing an Automatic Coverage Evaluation Metric","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada","keywords":"Automatic summarization; Argument (complex analysis); Key (lock); Metric (unit); Computer science; Point (geometry); Engineering; Artificial intelligence; Computer security; Mathematics; Operations management","score_opus":0.02062459436336553,"score_gpt":0.3120638485940622,"score_spread":0.29143925423069666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401043913","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12337349,0.0050902786,0.8499035,0.0011952302,0.0002938388,0.0008097384,0.0016777867,0.008192327,0.009463859],"genre_scores_gemma":[0.47902718,0.00087783515,0.5127627,0.00020141792,0.00018565293,0.00036265893,0.003362956,0.00095208175,0.002267463],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99086857,0.0039090705,0.00089737686,0.00097076443,0.0029943485,0.00035990708],"domain_scores_gemma":[0.9600812,0.025571002,0.0026203902,0.0029093407,0.0081115505,0.00070651213],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006286803,0.0014687666,0.0014606015,0.005948029,0.0008969966,0.0039617848,0.0019099033,0.0017490151,0.0033313294],"category_scores_gemma":[0.046726815,0.00059352285,0.0006939906,0.003055669,0.000909952,0.005359835,0.0028954688,0.0012815906,0.0016553428],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010359063,0.0004771344,0.009970592,0.0015613385,0.00018883687,0.0003156044,0.0014161103,0.028340707,0.042570908,0.014410513,0.01797178,0.88174057],"study_design_scores_gemma":[0.0002523388,0.0008223196,0.008581324,0.00045812692,0.00045390305,0.00060487865,0.00096338615,0.85711545,0.076272845,0.032618504,0.021695867,0.0001610657],"about_ca_topic_score_codex":0.002272107,"about_ca_topic_score_gemma":0.0041284203,"teacher_disagreement_score":0.006286803,"about_ca_system_score_codex":0.0009807523,"about_ca_system_score_gemma":0.0019483791,"threshold_uncertainty_score":0.033248186},"labels":[],"label_agreement":null},{"id":"W4401784090","doi":"10.1007/978-3-031-66694-0_17","title":"Citation Polarity Identification in Scientific Research Articles Using Deep Learning Methods","year":2024,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Identification (biology); Polarity (international relations); Citation; Computer science; Information retrieval; Data science; Library science; Chemistry; Botany; Biology; Biochemistry","score_opus":0.1929742432770527,"score_gpt":0.4777759264100746,"score_spread":0.2848016831330219,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401784090","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36422426,0.023768153,0.54376984,0.0050617238,0.0032251768,0.00024378159,0.0053194235,0.003599285,0.05078843],"genre_scores_gemma":[0.8438683,0.006158535,0.11995013,0.00041586367,0.0019414133,0.00013801378,0.005383411,0.00032484997,0.021819508],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992675,0.00014334089,0.00008784223,0.00011603007,0.00028375947,0.00010154673],"domain_scores_gemma":[0.9952429,0.002562647,0.00046115703,0.00020382645,0.0013650757,0.00016438004],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0011673945,0.00058222347,0.0005662536,0.006454863,0.00063452235,0.0024884725,0.0006076828,0.00086958037,0.002923554],"category_scores_gemma":[0.006184409,0.0002280796,0.0006659698,0.0047558183,0.00030251217,0.0021684193,0.0011035447,0.0013683711,0.0016015162],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030732437,0.00023768171,0.016809508,0.0006603744,0.00013697172,0.00024561718,0.0002322828,0.009144416,0.017113777,0.009543714,0.024473926,0.9210945],"study_design_scores_gemma":[0.000040727242,0.00012738514,0.019490352,0.0003673154,0.0003119367,0.00044439343,0.0003652858,0.8615877,0.02668743,0.059784684,0.030719813,0.000073037234],"about_ca_topic_score_codex":0.001420077,"about_ca_topic_score_gemma":0.0030991458,"teacher_disagreement_score":0.9988326,"about_ca_system_score_codex":0.00071952643,"about_ca_system_score_gemma":0.00081952824,"threshold_uncertainty_score":0.009780228},"labels":[],"label_agreement":null},{"id":"W4401943123","doi":"10.1109/tkde.2024.3443928","title":"PLBR: A Semi-Supervised Document Key Information Extraction via Pseudo-Labeling Bias Rectification","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Knowledge and Data Engineering","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Rectification; Key (lock); Artificial intelligence; Information extraction; Information retrieval; Pattern recognition (psychology)","score_opus":0.02764258680815602,"score_gpt":0.2946802442361714,"score_spread":0.2670376574280154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401943123","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0092713265,0.0010557283,0.9770454,0.00024683756,0.00013640976,0.00016059728,0.0006958573,0.010104133,0.00128381],"genre_scores_gemma":[0.09746048,0.0006994254,0.8852008,0.00048962666,0.00017867242,0.0003822497,0.0061391713,0.00094316463,0.008506403],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970421,0.0005295197,0.00018823845,0.001129413,0.00090891594,0.00020180292],"domain_scores_gemma":[0.9973399,0.0006121995,0.0003127884,0.0007418186,0.000901045,0.00009217606],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019155317,0.0020012285,0.0018156413,0.00247869,0.0008591077,0.0015685337,0.0029815217,0.0018327633,0.003500603],"category_scores_gemma":[0.0048242016,0.00077115727,0.0014196269,0.0024658074,0.0009163525,0.0035297074,0.002440058,0.0025709074,0.006194839],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040561473,0.00014222757,0.00069358794,0.000433892,0.00008440006,0.00013075404,0.00017984118,0.014233131,0.051666252,0.0030291788,0.018128006,0.91087306],"study_design_scores_gemma":[0.00012465996,0.00029228802,0.0015381118,0.00006642726,0.00010209304,0.000609313,0.00019214656,0.8560234,0.10066471,0.012367134,0.027896026,0.00012369575],"about_ca_topic_score_codex":0.0024447876,"about_ca_topic_score_gemma":0.0042050006,"teacher_disagreement_score":0.003500603,"about_ca_system_score_codex":0.00062594085,"about_ca_system_score_gemma":0.0016957113,"threshold_uncertainty_score":0.011710703},"labels":[],"label_agreement":null},{"id":"W4402422143","doi":"10.1515/lingvan-2023-0102","title":"Bibliographic bias and information-density sampling","year":2024,"lang":"en","type":"article","venue":"Linguistics Vanguard","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Marcus och Amalia Wallenbergs minnesfond","keywords":"Sampling bias; Sampling (signal processing); Statistics; Information retrieval; Computer science; Geography; Mathematics; Sample size determination; Telecommunications","score_opus":0.025144856216042854,"score_gpt":0.29749520760234294,"score_spread":0.2723503513863001,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402422143","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21595514,0.017126357,0.6698187,0.020177852,0.001079099,0.0026210404,0.0038725,0.00091013435,0.06843914],"genre_scores_gemma":[0.87963235,0.0029080939,0.10767119,0.0021333753,0.0008125696,0.0031534715,0.0013417656,0.00014920099,0.0021979462],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.61239564,0.2940697,0.023957502,0.01956209,0.04731511,0.0026999216],"domain_scores_gemma":[0.12950107,0.76372534,0.03291723,0.049906857,0.022819914,0.0011295564],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.23577125,0.0006860632,0.0027806433,0.024514185,0.0042841868,0.009370446,0.004950193,0.0025444792,0.006182659],"category_scores_gemma":[0.6992773,0.00117837,0.0010063632,0.040740408,0.011577865,0.012066471,0.007245197,0.0022082448,0.0012103791],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060813245,0.00016756891,0.11129969,0.0047150115,0.0010010082,0.0012273652,0.02224816,0.0072406484,0.0010592439,0.5850279,0.013092129,0.25231314],"study_design_scores_gemma":[0.00025433625,0.00018299649,0.04888704,0.0048572505,0.0005624277,0.0022688305,0.012221869,0.029361313,0.0029482103,0.83148074,0.066751465,0.00022343926],"about_ca_topic_score_codex":0.0069225836,"about_ca_topic_score_gemma":0.0059306785,"teacher_disagreement_score":0.9754858,"about_ca_system_score_codex":0.0058763176,"about_ca_system_score_gemma":0.0037091642,"threshold_uncertainty_score":0.9424301},"labels":[],"label_agreement":null},{"id":"W4402683031","doi":"10.18653/v1/2024.sighan-1.1","title":"Automatic Quote Attribution in Chinese Literary Works","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Attribution; Computer science; Authorship attribution; Natural language processing; Psychology; Social psychology","score_opus":0.006278342811272761,"score_gpt":0.2905591747985497,"score_spread":0.28428083198727694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402683031","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8323348,0.0029993637,0.13953061,0.0008194946,0.0005239016,0.00032594008,0.0035471094,0.007875556,0.012043252],"genre_scores_gemma":[0.9548055,0.00037818353,0.036533717,0.00006457012,0.00012217925,0.00005817724,0.0036939261,0.00019592715,0.004147837],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981786,0.0005083852,0.00016344938,0.0006395074,0.000361453,0.00014848524],"domain_scores_gemma":[0.9960418,0.0017646422,0.00052152027,0.0004287831,0.0010818494,0.00016153972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020811805,0.00072219496,0.000486516,0.0045201355,0.0012358581,0.0017821124,0.00091795437,0.00096514873,0.00258396],"category_scores_gemma":[0.008771701,0.00031506812,0.00051246275,0.0020052886,0.0006005049,0.0018881453,0.001852674,0.0009622905,0.0020048812],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014012958,0.00019199104,0.061591018,0.0017769577,0.00014114432,0.0025737872,0.011877018,0.008950016,0.086336724,0.008287025,0.017947415,0.79892564],"study_design_scores_gemma":[0.00008720974,0.00025835447,0.17392929,0.00041529842,0.00033569784,0.002082494,0.007887689,0.6173167,0.13018459,0.017275766,0.050000552,0.00022637533],"about_ca_topic_score_codex":0.003228626,"about_ca_topic_score_gemma":0.0029401511,"teacher_disagreement_score":0.0045201355,"about_ca_system_score_codex":0.00068564026,"about_ca_system_score_gemma":0.0007843729,"threshold_uncertainty_score":0.0110064745},"labels":[],"label_agreement":null},{"id":"W4402703397","doi":"10.1007/978-3-031-67317-7_9","title":"Machine Learning Based Extractive Text Summarization Using Document Aware and Document Unaware Features","year":2024,"lang":"en","type":"book-chapter","venue":"Studies in systems, decision and control","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Automatic summarization; Computer science; Information retrieval; Natural language processing; Artificial intelligence; World Wide Web","score_opus":0.02520959795395034,"score_gpt":0.3316281685645455,"score_spread":0.3064185706105952,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402703397","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02354778,0.002098975,0.95979893,0.00024538508,0.00039813144,0.00018112067,0.001176242,0.009803279,0.0027501844],"genre_scores_gemma":[0.15736459,0.001450584,0.81145245,0.00018421521,0.0006823596,0.0003090954,0.0076349042,0.00074687076,0.02017491],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994685,0.000084645515,0.00006646006,0.00014594181,0.00018582726,0.00004862795],"domain_scores_gemma":[0.99836844,0.0007691141,0.00015980801,0.00019946058,0.00046038165,0.00004278775],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005834931,0.0014918334,0.0014773837,0.001872526,0.0004990358,0.001558228,0.00090964336,0.00082626904,0.0044835294],"category_scores_gemma":[0.0018515359,0.000431256,0.0009923928,0.0019941279,0.00027630405,0.0018891087,0.00067705166,0.00120897,0.0044256966],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003790887,0.00014999062,0.00041963454,0.00036039742,0.00012300286,0.00018407914,0.00009673701,0.010858039,0.093046114,0.0014314887,0.009660702,0.8832906],"study_design_scores_gemma":[0.000094400144,0.0008711885,0.0049577868,0.000079549616,0.0005123561,0.00067355484,0.00022604724,0.7898494,0.17139582,0.0073933844,0.023839034,0.00010746918],"about_ca_topic_score_codex":0.00088930817,"about_ca_topic_score_gemma":0.0019276125,"teacher_disagreement_score":0.0044835294,"about_ca_system_score_codex":0.00031850018,"about_ca_system_score_gemma":0.0003992858,"threshold_uncertainty_score":0.014998853},"labels":[],"label_agreement":null},{"id":"W4402721984","doi":"10.1145/3670947.3670971","title":"TextVista: NLP-Enriched Time-Series Text Data Visualizations","year":2024,"lang":"en","type":"article","venue":"Graphics Interface","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Bruyère","funders":"Natural Sciences and Engineering Research Council of Canada; Universitas Brawijaya; Ontario Centre of Innovation","keywords":"Computer science; Natural language processing; Series (stratigraphy); Artificial intelligence; Time series; Visualization; Information retrieval; Machine learning","score_opus":0.03054556567332933,"score_gpt":0.3480504868164028,"score_spread":0.3175049211430735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402721984","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025923742,0.000616669,0.7744625,0.0016819224,0.00055454293,0.00091215817,0.030864986,0.15510365,0.009879778],"genre_scores_gemma":[0.16186947,0.00075370923,0.78684056,0.0005884898,0.0002709451,0.0028576015,0.020228988,0.018289028,0.008301261],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988067,0.00048187858,0.00011383147,0.0002325759,0.00031095688,0.000054038832],"domain_scores_gemma":[0.98946977,0.007627137,0.0006150354,0.0008257943,0.0011478371,0.00031449297],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032318027,0.0015967648,0.0007125095,0.0031534713,0.0007506029,0.0033235413,0.0016430591,0.0010993106,0.024270687],"category_scores_gemma":[0.017399421,0.0005228519,0.0010379303,0.0021686659,0.0006077793,0.0036983534,0.002169704,0.001665801,0.0039883023],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002777801,0.00045815122,0.0067814374,0.00486645,0.00039972775,0.0024762235,0.022553543,0.018470658,0.07337158,0.041304376,0.29310232,0.5334378],"study_design_scores_gemma":[0.0007190074,0.000616884,0.008476301,0.0014586404,0.00016902291,0.0013315771,0.0039645457,0.2801894,0.06551059,0.11933587,0.5177622,0.00046598277],"about_ca_topic_score_codex":0.0012815435,"about_ca_topic_score_gemma":0.0018936627,"teacher_disagreement_score":0.024270687,"about_ca_system_score_codex":0.00055780774,"about_ca_system_score_gemma":0.000880501,"threshold_uncertainty_score":0.08119351},"labels":[],"label_agreement":null},{"id":"W4402727755","doi":"10.1109/cvpr52733.2024.01054","title":"Discovering and Mitigating Visual Biases Through Keyword Explanation","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Data science; Information retrieval; Natural language processing","score_opus":0.024809341216219888,"score_gpt":0.3365943777522731,"score_spread":0.31178503653605316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402727755","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15028359,0.0011666488,0.8305259,0.0014086134,0.00015989973,0.0003097595,0.0012614741,0.011977914,0.0029062005],"genre_scores_gemma":[0.6769349,0.00040256258,0.31778678,0.00062271877,0.00015095736,0.00016507778,0.0019505325,0.00079760316,0.0011888674],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99501,0.0015814338,0.00045578575,0.0011175472,0.0014429464,0.0003922185],"domain_scores_gemma":[0.97535896,0.013705335,0.0033602193,0.003993199,0.003193344,0.0003889662],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056665055,0.0018723889,0.0013305867,0.003147861,0.0007562065,0.0023727415,0.0021917035,0.0020326434,0.0018387851],"category_scores_gemma":[0.03959372,0.00066904887,0.0013118202,0.0020049994,0.001348919,0.005957154,0.003520008,0.0018099189,0.0009866524],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014275636,0.00028031075,0.049594257,0.0012515958,0.00044635832,0.0008222778,0.00237491,0.0793521,0.059155196,0.01645984,0.01580252,0.7730331],"study_design_scores_gemma":[0.00009733943,0.0003065351,0.008911004,0.00017659628,0.00019821653,0.00066806114,0.00084293133,0.85056335,0.062033195,0.065733016,0.010311844,0.00015786239],"about_ca_topic_score_codex":0.0038841644,"about_ca_topic_score_gemma":0.0044181645,"teacher_disagreement_score":0.0056665055,"about_ca_system_score_codex":0.0015287873,"about_ca_system_score_gemma":0.0017867107,"threshold_uncertainty_score":0.029967725},"labels":[],"label_agreement":null},{"id":"W4402731657","doi":"10.23977/jaip.2024.070315","title":"Siamese Network-Based Text Similarity Algorithm Research","year":2024,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Practice","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Similarity (geometry); Computer science; Algorithm; Artificial intelligence","score_opus":0.1152017498653385,"score_gpt":0.4580224853169203,"score_spread":0.3428207354515818,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402731657","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021956066,0.0023643954,0.9685686,0.00053155556,0.00022732736,0.0000866178,0.00010535089,0.00069932296,0.0054608798],"genre_scores_gemma":[0.5141857,0.0032892004,0.46000177,0.00045584224,0.00053490914,0.00022209324,0.00083862926,0.00031479265,0.020157019],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990721,0.00019397643,0.00008098604,0.0002946894,0.0003085966,0.00004966424],"domain_scores_gemma":[0.99834204,0.00061583915,0.00015071307,0.00022202646,0.00059645536,0.00007293681],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011636474,0.00059953186,0.0009264801,0.0019885027,0.0004900215,0.0014730484,0.0016223043,0.0011645384,0.0035592658],"category_scores_gemma":[0.0055883033,0.0002911157,0.0007460724,0.0023003817,0.0008724536,0.0037124362,0.0009014721,0.0011971747,0.0011726252],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020131485,0.00023726614,0.0019167518,0.00032533915,0.00022189916,0.0001518987,0.0001857807,0.30934343,0.010061865,0.093121186,0.006338702,0.5778945],"study_design_scores_gemma":[0.000009768495,0.00004234917,0.00032477622,0.000007882115,0.000013533646,0.000044002587,0.0000135547,0.97825146,0.0021163877,0.016388113,0.0027793748,0.000008793008],"about_ca_topic_score_codex":0.006421442,"about_ca_topic_score_gemma":0.0045077316,"teacher_disagreement_score":0.006421442,"about_ca_system_score_codex":0.0013674032,"about_ca_system_score_gemma":0.0011658281,"threshold_uncertainty_score":0.012768149},"labels":[],"label_agreement":null},{"id":"W4402747380","doi":"10.23977/acss.2024.080603","title":"Research on Graph-based Text Summarization Extraction Algorithm","year":2024,"lang":"en","type":"article","venue":"Advances in Computer Signals and Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Graph; Text graph; Artificial intelligence; Natural language processing; Information retrieval; Algorithm; Theoretical computer science","score_opus":0.03971752346566083,"score_gpt":0.38430624482742454,"score_spread":0.3445887213617637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402747380","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006508163,0.0020828096,0.98577183,0.00029867652,0.0001510108,0.00017360166,0.00042948712,0.0031957359,0.0013886297],"genre_scores_gemma":[0.08385566,0.0027741601,0.9033732,0.0002844066,0.00030822912,0.00037256995,0.0035911635,0.00046453543,0.004976013],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99899894,0.00019188011,0.00013771623,0.00033176874,0.0002808556,0.00005894542],"domain_scores_gemma":[0.9983115,0.00063584617,0.00020545274,0.00014230929,0.00066469,0.000040296592],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070498086,0.0015120793,0.0011744938,0.004548095,0.0007327453,0.0012276798,0.0013050652,0.00096688606,0.0027869695],"category_scores_gemma":[0.0035086614,0.0004276086,0.0012822907,0.004130344,0.00045744955,0.0029282742,0.0005799601,0.00084864785,0.0019890754],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001530297,0.00007196701,0.00087952556,0.0009774887,0.00016544119,0.0002044265,0.00021884,0.03783578,0.033330053,0.011029119,0.010519778,0.9046146],"study_design_scores_gemma":[0.00010422297,0.0003323191,0.0024443143,0.00014277485,0.00034839366,0.0007088243,0.0002730047,0.85534763,0.05763292,0.034754828,0.047802035,0.00010879486],"about_ca_topic_score_codex":0.002682382,"about_ca_topic_score_gemma":0.0023209786,"teacher_disagreement_score":0.004548095,"about_ca_system_score_codex":0.0006385067,"about_ca_system_score_gemma":0.00095989363,"threshold_uncertainty_score":0.009323299},"labels":[],"label_agreement":null},{"id":"W4402761492","doi":"10.3390/systems12090380","title":"Learning to Score: A Coding System for Constructed Response Items via Interactive Clustering","year":2024,"lang":"en","type":"article","venue":"Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"National Social Science Fund of China; Fundo para o Desenvolvimento das Ciências e da Tecnologia; China Scholarship Council; Science and Technology Development Fund","keywords":"Cluster analysis; Coding (social sciences); Computer science; Psychology; Natural language processing; Artificial intelligence; Mathematics; Statistics","score_opus":0.015782493697655758,"score_gpt":0.29754658303594267,"score_spread":0.2817640893382869,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402761492","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006080092,0.000052894367,0.93533856,0.00016587184,0.00007893092,0.0014001385,0.0013250961,0.054148737,0.0014096962],"genre_scores_gemma":[0.04090064,0.000046245277,0.94889987,0.0001303476,0.00004024018,0.0022379153,0.003085834,0.0016359086,0.0030230058],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98927784,0.0038425736,0.00129967,0.002594492,0.0025969597,0.00038845086],"domain_scores_gemma":[0.97625494,0.007172976,0.001730037,0.004226464,0.009678971,0.0009365978],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011057634,0.0030120383,0.0014723389,0.0056436392,0.00130704,0.0021306034,0.0041707098,0.0017042867,0.013538274],"category_scores_gemma":[0.038964946,0.0009647623,0.0013246534,0.0026906764,0.0011078392,0.0036890863,0.005571673,0.0025330393,0.0121827135],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069597206,0.0004669591,0.0035166617,0.00037740543,0.00009866733,0.00017650251,0.0018659723,0.007930658,0.01448568,0.0049690944,0.044933572,0.9204829],"study_design_scores_gemma":[0.00035398157,0.0005551111,0.006624476,0.0002062182,0.00009484368,0.0003904067,0.0011612743,0.8466845,0.051529475,0.029880209,0.062132172,0.00038729503],"about_ca_topic_score_codex":0.004699548,"about_ca_topic_score_gemma":0.0058870846,"teacher_disagreement_score":0.013538274,"about_ca_system_score_codex":0.0020373361,"about_ca_system_score_gemma":0.0026968943,"threshold_uncertainty_score":0.05847901},"labels":[],"label_agreement":null},{"id":"W4402978475","doi":"10.1109/tits.2024.3462951","title":"eMARLIN+: Addressing Partial Observability to Promote Traffic Signal Coordination by Leveraging Historical Information","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Intelligent Transportation Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Observability; SIGNAL (programming language); Computer science; Transport engineering; Computer security; Engineering; Mathematics","score_opus":0.04013403747008683,"score_gpt":0.28761673632990536,"score_spread":0.24748269885981855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402978475","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034056924,0.0005411112,0.93963635,0.00041383042,0.00032971406,0.00010397572,0.0011826757,0.017635742,0.00609969],"genre_scores_gemma":[0.5414648,0.00045957416,0.43956026,0.00035380502,0.00026367212,0.00015989013,0.0036422322,0.0012318281,0.01286395],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99929833,0.00018751911,0.00004054221,0.00022167142,0.00018584683,0.00006606324],"domain_scores_gemma":[0.9981864,0.0007922126,0.00015358327,0.00050716137,0.000266184,0.00009433889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012778577,0.0010852489,0.0007894359,0.0009563244,0.00049962074,0.0010325054,0.0017556387,0.0007718156,0.0060238075],"category_scores_gemma":[0.0043441923,0.00039574283,0.00047124736,0.0007080265,0.0004390399,0.002470258,0.002010809,0.0011451844,0.0013768361],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011620888,0.0005094862,0.0041361707,0.00057541474,0.0001909058,0.00034621358,0.00026784764,0.20138666,0.024505692,0.025054937,0.04080664,0.701058],"study_design_scores_gemma":[0.000060275794,0.00022083089,0.0005168631,0.000022150614,0.000054130614,0.00011688623,0.00003849796,0.9588585,0.014529133,0.0125433635,0.013011143,0.000028157252],"about_ca_topic_score_codex":0.0017026407,"about_ca_topic_score_gemma":0.0050657485,"teacher_disagreement_score":0.0060238075,"about_ca_system_score_codex":0.00029451118,"about_ca_system_score_gemma":0.0009981404,"threshold_uncertainty_score":0.020151615},"labels":[],"label_agreement":null},{"id":"W4403134937","doi":"10.1080/08839514.2024.2403904","title":"Integration of Neural Embeddings and Probabilistic Models in Topic Modeling","year":2024,"lang":"en","type":"article","venue":"Applied Artificial Intelligence","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Probabilistic logic; Artificial neural network; Statistical model; Artificial intelligence; Machine learning; Data mining; Data science; Theoretical computer science","score_opus":0.06095076820110108,"score_gpt":0.31866020918327603,"score_spread":0.25770944098217496,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403134937","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010631175,0.00087850384,0.98673767,0.00041881684,0.000057936868,0.000025916876,0.00011390969,0.0003442584,0.0007918978],"genre_scores_gemma":[0.57717836,0.0028607172,0.41378257,0.00037699364,0.00055128225,0.00027461207,0.0011919644,0.00027122,0.00351228],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982303,0.0008850599,0.00013532198,0.0003875955,0.00027808352,0.00008374254],"domain_scores_gemma":[0.99356836,0.004602468,0.0004352717,0.00075212226,0.0005070139,0.00013471054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032495107,0.0011233841,0.0010285146,0.0020666115,0.00048369484,0.0022835059,0.001526976,0.0013821871,0.0012801485],"category_scores_gemma":[0.014629246,0.00069063465,0.0012437429,0.0024375303,0.0009936624,0.007302726,0.0019694048,0.0029309664,0.00079117896],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018851043,0.00018280496,0.004373742,0.0003543922,0.00030453736,0.00014009794,0.0007409954,0.4615845,0.0037205834,0.13752612,0.0039553577,0.3869283],"study_design_scores_gemma":[0.000004976025,0.000019460107,0.00023281206,0.000017299695,0.000017650964,0.000032128966,0.000027158785,0.9382377,0.00041391808,0.05986486,0.0011188992,0.000013170369],"about_ca_topic_score_codex":0.0027041594,"about_ca_topic_score_gemma":0.0045545157,"teacher_disagreement_score":0.0032495107,"about_ca_system_score_codex":0.001018255,"about_ca_system_score_gemma":0.0007618552,"threshold_uncertainty_score":0.01718527},"labels":[],"label_agreement":null},{"id":"W4403181741","doi":"10.1007/978-3-031-63821-3_6","title":"Natural Language Processing for Emotion Recognition and Analysis","year":2024,"lang":"en","type":"book-chapter","venue":"The Springer series in applied machine learning","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Computer science; Natural language processing; Speech recognition; Psychology; Natural (archaeology); Communication; Biology; Paleontology","score_opus":0.01109883442005598,"score_gpt":0.2515212515609263,"score_spread":0.24042241714087031,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403181741","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037206842,0.028978953,0.865152,0.0029900959,0.0028973692,0.00022747653,0.002688107,0.009473333,0.08387195],"genre_scores_gemma":[0.06297269,0.022109663,0.6858638,0.0016660767,0.0019954157,0.0005579322,0.00814951,0.0021932917,0.21449168],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99969435,0.000065152904,0.000026412219,0.000083126106,0.00011261945,0.000018380868],"domain_scores_gemma":[0.99962103,0.000171269,0.000019590458,0.00006568928,0.000109950844,0.000012461053],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00044987645,0.00080152735,0.00052062224,0.0007814186,0.00031042076,0.0018878819,0.0010113968,0.0006297196,0.031228125],"category_scores_gemma":[0.0010721362,0.00023371296,0.0005322815,0.0011022881,0.0004785271,0.0024597405,0.0007804738,0.001008114,0.018203897],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000035126253,0.000032110354,0.00009216811,0.0004931071,0.000023250162,0.00009034664,0.00014947941,0.0008582943,0.015102644,0.028316252,0.13385615,0.8209511],"study_design_scores_gemma":[0.00001623878,0.00006668269,0.0012584566,0.0003192209,0.000048558144,0.0007221166,0.0002869163,0.046310212,0.020675903,0.14476867,0.7854693,0.000057777623],"about_ca_topic_score_codex":0.0009178577,"about_ca_topic_score_gemma":0.0013321013,"teacher_disagreement_score":0.031228125,"about_ca_system_score_codex":0.00036680003,"about_ca_system_score_gemma":0.0003740983,"threshold_uncertainty_score":0.104468465},"labels":[],"label_agreement":null},{"id":"W4403210328","doi":"10.1109/iri62200.2024.00029","title":"Accelerating Relational Keyword Queries With Embedded Predictive Neural Networks","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Artificial neural network; Keyword search; Relational database; Information retrieval; Artificial intelligence; Data mining","score_opus":0.016837767891135235,"score_gpt":0.26378782246279087,"score_spread":0.24695005457165564,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403210328","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33008665,0.0024441986,0.62781113,0.0009365689,0.00023015808,0.00023853974,0.0011631802,0.029011657,0.008077946],"genre_scores_gemma":[0.8416163,0.0004138441,0.151867,0.00024020132,0.00006984413,0.00009129203,0.0015328934,0.00022770623,0.003940853],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995146,0.00006481809,0.000041322794,0.00013305695,0.00017487336,0.00007132587],"domain_scores_gemma":[0.99836594,0.0008461114,0.00013294232,0.00022173615,0.00038652078,0.000046744382],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006343334,0.00075982587,0.00066950166,0.00074191974,0.00034538688,0.0010757861,0.0015620117,0.00070609263,0.0027114837],"category_scores_gemma":[0.004714832,0.0003379361,0.0003019261,0.0012966711,0.00034745247,0.0029133388,0.0008459533,0.000912193,0.0009956999],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010660527,0.00063274277,0.0043595717,0.00029028923,0.0001306674,0.00028494856,0.00019026268,0.3561033,0.048191402,0.0059038554,0.0137513345,0.5690956],"study_design_scores_gemma":[0.000012878063,0.00004929064,0.00021569485,0.0000034634108,0.000011340761,0.000026577123,0.000022084287,0.9899824,0.0076328577,0.0015348075,0.0005022978,0.000006303578],"about_ca_topic_score_codex":0.01477563,"about_ca_topic_score_gemma":0.016803961,"teacher_disagreement_score":0.01477563,"about_ca_system_score_codex":0.0012880174,"about_ca_system_score_gemma":0.0008995728,"threshold_uncertainty_score":0.029379249},"labels":[],"label_agreement":null},{"id":"W4403432858","doi":"10.1002/pra2.1017","title":"Exploratory Search in Digital Humanities: A Study of Visual Keyword/Result Linking","year":2024,"lang":"en","type":"article","venue":"Proceedings of the Association for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Digital humanities; Information retrieval; Computer science; Humanities; World Wide Web; Art","score_opus":0.01574173510886453,"score_gpt":0.29249908303467237,"score_spread":0.27675734792580786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403432858","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9327306,0.00082027103,0.043476902,0.0005598609,0.00004233952,0.0012561379,0.00037586867,0.0014728062,0.0192652],"genre_scores_gemma":[0.9579246,0.0002689442,0.039156344,0.00020988325,0.000026989623,0.0006879827,0.00029929387,0.00029841607,0.0011275677],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97014576,0.022828687,0.0016987448,0.001421833,0.0033886184,0.00051621156],"domain_scores_gemma":[0.4299852,0.5367267,0.014421521,0.010390889,0.006903356,0.0015723484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02941949,0.0006551351,0.0009540552,0.006407229,0.0014512312,0.006826557,0.0021330784,0.001365047,0.004513005],"category_scores_gemma":[0.27312148,0.00063157355,0.0007091789,0.0058906274,0.0026968739,0.010945228,0.004894025,0.0015847922,0.0006531927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008224884,0.009067293,0.21357945,0.0067844195,0.0008683725,0.0022205375,0.15583715,0.03150042,0.031331677,0.044126328,0.0063874894,0.49007198],"study_design_scores_gemma":[0.0022365218,0.012568392,0.24308065,0.0042492272,0.001836321,0.005803454,0.108503,0.42456147,0.052208226,0.085460775,0.05835765,0.0011343873],"about_ca_topic_score_codex":0.0038142877,"about_ca_topic_score_gemma":0.0018362235,"teacher_disagreement_score":0.02941949,"about_ca_system_score_codex":0.0017009604,"about_ca_system_score_gemma":0.0016386486,"threshold_uncertainty_score":0.15558702},"labels":[],"label_agreement":null},{"id":"W4403494021","doi":"10.1561/116.20240044","title":"Automatic Medical Report Generation: Methods and Applications","year":2024,"lang":"en","type":"article","venue":"APSIPA Transactions on Signal and Information Processing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science","score_opus":0.015513525712463184,"score_gpt":0.3377518521077463,"score_spread":0.3222383263952831,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403494021","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007707759,0.03406786,0.9277976,0.005037674,0.0008742052,0.00075858016,0.007041395,0.008608428,0.008106416],"genre_scores_gemma":[0.095855586,0.031341527,0.8523957,0.0011328926,0.0015186951,0.00087633776,0.011139605,0.0007088879,0.005030765],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954904,0.0016434041,0.00048439085,0.00078389724,0.001484859,0.000113074595],"domain_scores_gemma":[0.9856418,0.008872521,0.0009652009,0.0016334503,0.00266065,0.00022635741],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074173477,0.0014488664,0.00091468386,0.0050017266,0.00056502246,0.0031926997,0.002491776,0.0020826734,0.0067768306],"category_scores_gemma":[0.02716711,0.0005862916,0.0011305318,0.005041217,0.00077059824,0.0023809976,0.0015390642,0.0015861298,0.006509784],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017034737,0.00009679211,0.0025356195,0.0011979195,0.00010435344,0.00020199081,0.00015526998,0.011767046,0.0035955396,0.0070254696,0.03675477,0.9363949],"study_design_scores_gemma":[0.00024141111,0.00040178144,0.009416612,0.0018617206,0.0003099883,0.003451526,0.0006230823,0.47993147,0.038548894,0.09851724,0.36632162,0.0003746544],"about_ca_topic_score_codex":0.0025004416,"about_ca_topic_score_gemma":0.0020368602,"teacher_disagreement_score":0.0074173477,"about_ca_system_score_codex":0.0010085655,"about_ca_system_score_gemma":0.001636349,"threshold_uncertainty_score":0.039227188},"labels":[],"label_agreement":null},{"id":"W4404401204","doi":"10.54195/irrj.19910","title":"Annotative Indexing","year":2025,"lang":"en","type":"preprint","venue":"Information Retrieval Research","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Tsinghua University; University of Glasgow","keywords":"Search engine indexing; Computer science; Information retrieval","score_opus":0.08469652661856103,"score_gpt":0.4552504328181435,"score_spread":0.37055390619958245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404401204","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0041465717,0.0014574096,0.9173707,0.0014130677,0.0010325824,0.00047741365,0.00625408,0.013298269,0.05454989],"genre_scores_gemma":[0.09355291,0.0029942752,0.8080494,0.0027765466,0.0011247233,0.0010357337,0.029231992,0.008889377,0.052345082],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9873761,0.0021894027,0.0015750342,0.002098649,0.0057292213,0.0010316292],"domain_scores_gemma":[0.9675607,0.0055053495,0.0014110551,0.018128583,0.006579311,0.0008149689],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073481216,0.0013001077,0.0019263688,0.0067471433,0.0044933944,0.011852283,0.005488054,0.002007262,0.025749093],"category_scores_gemma":[0.028965874,0.0010799057,0.0018586222,0.011126319,0.0031114907,0.021603396,0.012309467,0.0028299824,0.016715884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004227409,0.00017040681,0.0021988517,0.0006948376,0.00009587537,0.00040923583,0.0021408843,0.0039795483,0.006805926,0.4884667,0.13365498,0.36095995],"study_design_scores_gemma":[0.000029629364,0.00005661768,0.0003948597,0.00031238954,0.000058412243,0.0005547975,0.00069332495,0.015133399,0.010910379,0.24874309,0.72299063,0.00012251048],"about_ca_topic_score_codex":0.0061669163,"about_ca_topic_score_gemma":0.0059830802,"teacher_disagreement_score":0.025749093,"about_ca_system_score_codex":0.0026948226,"about_ca_system_score_gemma":0.0050669215,"threshold_uncertainty_score":0.08613926},"labels":[],"label_agreement":null},{"id":"W4404783428","doi":"10.18653/v1/2024.conll-1.1","title":"Words That Stick: Using Keyword Cohesion to Improve Text Segmentation","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Cohesion (chemistry); Segmentation; Natural language processing; Artificial intelligence; Information retrieval; Keyword search","score_opus":0.0256191780195542,"score_gpt":0.3269138498247393,"score_spread":0.3012946718051851,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404783428","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37824452,0.004648749,0.560216,0.0007466376,0.00088850193,0.0004393972,0.0028465297,0.043586586,0.008383079],"genre_scores_gemma":[0.66516453,0.0004943707,0.31556705,0.00037328378,0.00036632956,0.00022962235,0.0073703,0.003763796,0.0066706575],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99881977,0.00019606974,0.00010941238,0.0005140818,0.0002451737,0.00011549882],"domain_scores_gemma":[0.996485,0.0011995744,0.00051925005,0.00080355594,0.0007134468,0.00027913917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009155802,0.0015213643,0.0010665958,0.0035111748,0.0008695604,0.0015557115,0.0010070715,0.0009391942,0.005657888],"category_scores_gemma":[0.007017183,0.00047898988,0.00079217856,0.0028055725,0.0007426419,0.0044913064,0.0025782015,0.0013350842,0.0044262805],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017127643,0.00031954766,0.0067479946,0.00054454774,0.00018393747,0.00031856835,0.0017690326,0.009413738,0.10398117,0.0030205126,0.018820982,0.85316724],"study_design_scores_gemma":[0.0006042719,0.0034678252,0.03530513,0.0002616526,0.0009824262,0.0014044417,0.003608269,0.5619767,0.2816775,0.04268206,0.06769194,0.00033776334],"about_ca_topic_score_codex":0.0024385764,"about_ca_topic_score_gemma":0.0037171766,"teacher_disagreement_score":0.005657888,"about_ca_system_score_codex":0.00048621788,"about_ca_system_score_gemma":0.00081969926,"threshold_uncertainty_score":0.018927515},"labels":[],"label_agreement":null},{"id":"W4404851827","doi":"10.1080/01973533.2024.2433720","title":"A Bibliometric Review of Natural Language Processing Applications in Psychology from 1991 to 2023","year":2024,"lang":"en","type":"review","venue":"Basic and Applied Social Psychology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of Alberta; University of British Columbia","funders":"","keywords":"Psychology; Natural (archaeology); Social psychology; Cognitive psychology; Applied psychology","score_opus":0.049524424754623125,"score_gpt":0.46184174114458415,"score_spread":0.41231731638996105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404851827","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010360819,0.99282587,0.00036081683,0.0012907591,0.00032026286,0.00006184298,0.0016773143,0.000041520318,0.0023854545],"genre_scores_gemma":[0.004956123,0.9924396,0.00055000215,0.0003539234,0.00026391653,0.00007938402,0.0010460325,0.0000137634415,0.00029725337],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99189985,0.0018679415,0.002336308,0.0007302728,0.0029228537,0.0002428411],"domain_scores_gemma":[0.9546736,0.026663827,0.005370218,0.0008680422,0.011707807,0.00071654236],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.009850494,0.0013801825,0.0035997324,0.09351635,0.0010153557,0.0032541333,0.0016141721,0.0011804552,0.006785685],"category_scores_gemma":[0.03860897,0.0010714807,0.0029047346,0.1160045,0.0011336214,0.0036675537,0.0021000458,0.0012329222,0.0019213788],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016433871,0.000052159514,0.0042384486,0.29559377,0.0025664314,0.0002969166,0.0004491325,0.0004926319,0.0005508063,0.0037042107,0.061441846,0.6304493],"study_design_scores_gemma":[0.00007545247,0.00015650807,0.033094347,0.28499565,0.010599662,0.0010576367,0.0007357437,0.00045059304,0.0008429401,0.0032383257,0.6646243,0.0001288287],"about_ca_topic_score_codex":0.008166554,"about_ca_topic_score_gemma":0.016082678,"teacher_disagreement_score":0.90648365,"about_ca_system_score_codex":0.004275814,"about_ca_system_score_gemma":0.012098368,"threshold_uncertainty_score":0.052094996},"labels":[],"label_agreement":null},{"id":"W4404940533","doi":"10.1007/s42113-024-00214-8","title":"Lessons for Theory from Scientific Domains Where Evidence is Sparse or Indirect","year":2024,"lang":"en","type":"article","venue":"Computational Brain & Behavior","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"Natural Sciences and Engineering Research Council of Canada; Nederlandse Organisatie voor Wetenschappelijk Onderzoek; James S. McDonnell Foundation","keywords":"Epistemology; Computer science; Econometrics; Data science; Mathematics; Philosophy","score_opus":0.0854073085012884,"score_gpt":0.38960454482013085,"score_spread":0.30419723631884243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404940533","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01574307,0.04992844,0.45275643,0.4155754,0.003696284,0.0003291588,0.0007060882,0.000369459,0.060895562],"genre_scores_gemma":[0.59834063,0.035122678,0.30275705,0.04745003,0.009075991,0.0014990674,0.0007750789,0.00043143833,0.004548073],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.96555805,0.0218679,0.0023354639,0.0037970843,0.0055995784,0.000841987],"domain_scores_gemma":[0.74311775,0.22015022,0.005509893,0.017665982,0.010139207,0.0034169576],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06108921,0.0018620791,0.0034310613,0.009796269,0.00470222,0.01474665,0.0054210066,0.0071433196,0.007982608],"category_scores_gemma":[0.13854286,0.001482378,0.0021339192,0.0054639312,0.060346063,0.04503905,0.012148148,0.017740404,0.0013841835],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027900434,0.000023490802,0.0007590376,0.00062970765,0.000116660514,0.00024276073,0.0017565499,0.0014776787,0.000074131734,0.972659,0.004598526,0.017634477],"study_design_scores_gemma":[0.000008199388,0.0000029842934,0.00006148262,0.00016539864,0.000005457418,0.00003106388,0.00018550303,0.00045386783,0.000025307992,0.99494827,0.0041058916,0.0000064673477],"about_ca_topic_score_codex":0.00348427,"about_ca_topic_score_gemma":0.004152405,"teacher_disagreement_score":0.06108921,"about_ca_system_score_codex":0.0066595594,"about_ca_system_score_gemma":0.0084695155,"threshold_uncertainty_score":0.32307446},"labels":[],"label_agreement":null},{"id":"W4405034837","doi":"10.48550/arxiv.2412.01621","title":"NYT-Connections: A Deceptively Simple Text Classification Task that Stumps System-1 Thinkers","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Simple (philosophy); Task (project management); Computer science; Linguistics; Natural language processing; Psychology; Cognitive psychology; Epistemology; Philosophy; Economics; Management","score_opus":0.08503281301773807,"score_gpt":0.22324432976329897,"score_spread":0.1382115167455609,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405034837","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7470392,0.0016781472,0.122085445,0.0036023392,0.001505153,0.0015128494,0.01915306,0.059729256,0.043694478],"genre_scores_gemma":[0.81027955,0.00032323142,0.14766185,0.0010283643,0.00019684299,0.00083348283,0.022555234,0.0025569112,0.014564472],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99756986,0.00084050227,0.00023766152,0.0006396571,0.0005587345,0.00015353425],"domain_scores_gemma":[0.9802256,0.014824851,0.0009307107,0.0023955333,0.0009692467,0.0006539983],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025199736,0.0020396216,0.0007465656,0.0011499346,0.0010083431,0.0019175301,0.0024189695,0.0023265867,0.009192308],"category_scores_gemma":[0.0349698,0.00047205866,0.00074337656,0.00085821055,0.0012005412,0.005011485,0.0027091124,0.0025683285,0.0041589565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0052033877,0.0019601588,0.030763242,0.0033343085,0.0006184004,0.0012290545,0.002808936,0.07930251,0.0244183,0.025430106,0.2579723,0.5669593],"study_design_scores_gemma":[0.0007905992,0.0016192722,0.009817233,0.00021713815,0.00012400142,0.0007900727,0.0009632259,0.8367217,0.026845748,0.04706456,0.07485582,0.00019063709],"about_ca_topic_score_codex":0.0071928343,"about_ca_topic_score_gemma":0.013661204,"teacher_disagreement_score":0.009192308,"about_ca_system_score_codex":0.0011986707,"about_ca_system_score_gemma":0.0018482114,"threshold_uncertainty_score":0.030751348},"labels":[],"label_agreement":null},{"id":"W4405918232","doi":"10.1002/asi.24977","title":"A study of drag‐and‐drop query refinement and query history visualization for mobile exploratory search","year":2024,"lang":"en","type":"article","venue":"Journal of the Association for Information Science and Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Information retrieval; Visualization; Drag; Data mining; Physics; Mechanics","score_opus":0.018390033844011486,"score_gpt":0.31747201513829915,"score_spread":0.29908198129428765,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405918232","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9921801,0.00016928035,0.0059416196,0.00006596686,0.000011721811,0.00029709062,0.00007531834,0.00030664552,0.0009522357],"genre_scores_gemma":[0.9830735,0.00009774362,0.01568879,0.00006849997,0.000013782159,0.0003757391,0.000115779556,0.00006540503,0.00050073635],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99729437,0.0017195033,0.00016803447,0.00025411724,0.00040568534,0.00015837884],"domain_scores_gemma":[0.92752975,0.06571132,0.001861322,0.002520295,0.0015847964,0.0007925138],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004274075,0.0006956666,0.000899461,0.0008340461,0.0007933597,0.0018152084,0.0012213268,0.00095196493,0.0031803723],"category_scores_gemma":[0.04142116,0.00056858704,0.00041076003,0.0006216022,0.0007432993,0.0020136214,0.0012631471,0.000879452,0.0003745854],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.011543173,0.012614204,0.10368365,0.008199445,0.00050209253,0.0045760577,0.19221255,0.009276001,0.31277192,0.004586167,0.007283516,0.33275124],"study_design_scores_gemma":[0.0055462653,0.0672803,0.4237594,0.0021919112,0.0018162833,0.0076927803,0.07794225,0.21424794,0.14030962,0.0061237803,0.051917866,0.0011716231],"about_ca_topic_score_codex":0.0016250104,"about_ca_topic_score_gemma":0.0016835837,"teacher_disagreement_score":0.004274075,"about_ca_system_score_codex":0.00048269948,"about_ca_system_score_gemma":0.00051981374,"threshold_uncertainty_score":0.02260369},"labels":[],"label_agreement":null},{"id":"W4406121435","doi":"10.1016/j.pharmr.2025.100037","title":"Anticoagulants: From chance discovery to structure-based design","year":2025,"lang":"en","type":"review","venue":"Pharmacological Reviews","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; Thrombosis and Atherosclerosis Research Institute","funders":"","keywords":"Drug discovery; Computational biology; Chemistry; Biology; Biochemistry","score_opus":0.149436100269024,"score_gpt":0.4509059683486288,"score_spread":0.30146986807960474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406121435","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00039817998,0.99051636,0.006299873,0.0010088003,0.00029362555,0.00003799947,0.00008387127,0.0000840734,0.0012772448],"genre_scores_gemma":[0.0070251552,0.9823307,0.008068991,0.0009285337,0.0006082734,0.000080354825,0.00013481178,0.000017926113,0.0008053577],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.999361,0.0002304956,0.00007320543,0.00010970546,0.00019751729,0.000028110682],"domain_scores_gemma":[0.99741274,0.002101033,0.00023272444,0.00007701558,0.0001295566,0.000046916877],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017362962,0.0011441185,0.002756398,0.002024336,0.00024461406,0.0014758777,0.0015914714,0.0010405653,0.0057871477],"category_scores_gemma":[0.004843789,0.0003997392,0.0015087252,0.001507168,0.0009067666,0.0016011914,0.00077815127,0.0021472494,0.0012995828],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013474283,0.00009527882,0.00023854428,0.01683648,0.00041995145,0.000102506056,0.000039443963,0.0018332951,0.0010351137,0.016188588,0.014264767,0.94881135],"study_design_scores_gemma":[0.0007033405,0.0012014483,0.0017524452,0.024270207,0.002363351,0.0019091048,0.00010665921,0.010449369,0.0059642233,0.14933331,0.8017517,0.0001948536],"about_ca_topic_score_codex":0.0006999399,"about_ca_topic_score_gemma":0.0011430569,"teacher_disagreement_score":0.0057871477,"about_ca_system_score_codex":0.0005930264,"about_ca_system_score_gemma":0.001274844,"threshold_uncertainty_score":0.019359887},"labels":[],"label_agreement":null},{"id":"W4406376912","doi":"10.1111/jedm.12424","title":"Using Multilabel Neural Network to Score High‐Dimensional Assessments for Different Use Foci: An Example with College Major Preference Assessment","year":2025,"lang":"en","type":"article","venue":"Journal of Educational Measurement","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Preference; Artificial neural network; Artificial intelligence; Psychology; Machine learning; Computer science; Evaluation methods; Statistics; Mathematics; Reliability engineering","score_opus":0.23328784592740195,"score_gpt":0.3977295805474502,"score_spread":0.16444173462004827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406376912","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.842434,0.00030276133,0.1450103,0.0006970361,0.00015738406,0.0002564804,0.00096690434,0.0015666175,0.008608528],"genre_scores_gemma":[0.93482476,0.000057874888,0.06269716,0.00007884552,0.000018044175,0.00009417657,0.00047728757,0.000035678746,0.001716202],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985649,0.0007206277,0.00010101139,0.00023456126,0.00030073914,0.00007819398],"domain_scores_gemma":[0.9949344,0.0030587225,0.00027545623,0.00043750956,0.0010793246,0.0002147515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027685766,0.0006699571,0.00040485666,0.0012165074,0.000385263,0.0008630711,0.00037684833,0.0005529083,0.002456608],"category_scores_gemma":[0.010534732,0.000107431995,0.00034277805,0.0010204382,0.00024241237,0.0008844138,0.00080799777,0.00076910935,0.0006899665],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008671696,0.0006704953,0.13475047,0.00019840058,0.0001957039,0.00034978343,0.0007296992,0.03269896,0.013855819,0.0017899547,0.007876349,0.8060172],"study_design_scores_gemma":[0.0000643161,0.000513527,0.099749275,0.000068604975,0.00009061274,0.00025462534,0.0010867413,0.8698592,0.0152359065,0.008103236,0.0048785596,0.00009544262],"about_ca_topic_score_codex":0.006222336,"about_ca_topic_score_gemma":0.014663607,"teacher_disagreement_score":0.006222336,"about_ca_system_score_codex":0.0006062814,"about_ca_system_score_gemma":0.00046953405,"threshold_uncertainty_score":0.014641821},"labels":[],"label_agreement":null},{"id":"W4406526032","doi":"10.1007/978-3-031-83188-1","title":"Multimodal Learning toward Recommendation","year":2025,"lang":"en","type":"book","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National University of Singapore; Institute for Catastrophic Loss Reduction; Harbin Institute of Technology; Indian Council of Medical Research; Shandong University","keywords":"Computer science; Psychology","score_opus":0.016548706271586278,"score_gpt":0.29154211845566214,"score_spread":0.27499341218407586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406526032","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008665976,0.088522986,0.35624853,0.006759749,0.0025249897,0.00017507638,0.0008209841,0.0047649755,0.5315168],"genre_scores_gemma":[0.10440209,0.07210596,0.2279602,0.0024259363,0.001923184,0.00028607517,0.0015464297,0.0007303626,0.5886198],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9997156,0.00007128194,0.00000734903,0.000050130584,0.00012945522,0.000026150889],"domain_scores_gemma":[0.99967647,0.00017239648,0.000010727574,0.000044627093,0.00007040446,0.000025324513],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00041902607,0.00074485154,0.00039395547,0.00072483235,0.000562526,0.0021573901,0.0007862849,0.0008253956,0.044748917],"category_scores_gemma":[0.0018725226,0.00025106827,0.0004669035,0.0015315579,0.00047293666,0.0033449873,0.0011970823,0.0015926098,0.015849473],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000068967835,0.000069507965,0.00018062328,0.00033691298,0.00002413783,0.00007454712,0.00020851484,0.0032553966,0.002877557,0.059086856,0.17128566,0.76253134],"study_design_scores_gemma":[0.000021050873,0.00014103448,0.0007358195,0.00057919614,0.00004726114,0.0006500592,0.00040228,0.034177106,0.004987189,0.117540725,0.8406687,0.00004972087],"about_ca_topic_score_codex":0.002152069,"about_ca_topic_score_gemma":0.0031764717,"teacher_disagreement_score":0.044748917,"about_ca_system_score_codex":0.00074884173,"about_ca_system_score_gemma":0.00043323546,"threshold_uncertainty_score":0.1497001},"labels":[],"label_agreement":null},{"id":"W4406800158","doi":"10.1007/978-3-031-78554-2_14","title":"A Real-Time Sentiment Feedback System: Binary Categorization and Context Understanding Based on Product Reviews","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Categorization; Context (archaeology); Product (mathematics); Binary number; Sentiment analysis; Artificial intelligence; Information retrieval; Mathematics; Arithmetic","score_opus":0.024243133494449783,"score_gpt":0.26564189855210996,"score_spread":0.24139876505766017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406800158","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3166757,0.0036642798,0.5195868,0.0009774573,0.0013834592,0.0021117686,0.011916222,0.13284703,0.0108372895],"genre_scores_gemma":[0.49225014,0.00069176074,0.47874108,0.00057163223,0.00073813327,0.0009831265,0.012550676,0.00108473,0.01238877],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991322,0.00015323567,0.00007425145,0.00023139418,0.000357617,0.00005127429],"domain_scores_gemma":[0.99776435,0.0006966702,0.00022866116,0.00012698275,0.0010167289,0.00016664526],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014170463,0.0012936082,0.0015809572,0.0028854336,0.0003862806,0.000997387,0.0010477982,0.0009819289,0.0055520176],"category_scores_gemma":[0.003146561,0.0004082458,0.0004812782,0.0013549337,0.00009576321,0.001343596,0.0007192032,0.0005494797,0.006233046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013731491,0.0007898349,0.0039587994,0.0006091058,0.00017255265,0.00018219996,0.00013166343,0.0013703457,0.120070845,0.00021641939,0.041168004,0.8299572],"study_design_scores_gemma":[0.00052671577,0.0017501473,0.035949074,0.00008557455,0.00050728716,0.00062968186,0.00027513143,0.81502014,0.12127537,0.0013425784,0.022406073,0.00023222608],"about_ca_topic_score_codex":0.0017838684,"about_ca_topic_score_gemma":0.0031154258,"teacher_disagreement_score":0.0055520176,"about_ca_system_score_codex":0.00039070458,"about_ca_system_score_gemma":0.0004313742,"threshold_uncertainty_score":0.018573344},"labels":[],"label_agreement":null},{"id":"W4407219662","doi":"10.1075/ml.24006.wes","title":"Orthographic uncertainty","year":2024,"lang":"en","type":"article","venue":"The Mental Lexicon","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.0113426503510872,"score_gpt":0.281487742315883,"score_spread":0.2701450919647958,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407219662","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.73722696,0.00268987,0.18639219,0.0011503015,0.00020051774,0.00011848669,0.0037604664,0.000532044,0.067929074],"genre_scores_gemma":[0.99039114,0.00028032332,0.007329033,0.00007201723,0.00008604239,0.000028013292,0.00058484386,0.000030406489,0.0011981154],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99774504,0.0005445739,0.00024836793,0.00043040366,0.0009409545,0.000090704234],"domain_scores_gemma":[0.98065096,0.011441889,0.0034079484,0.0020204843,0.0021131926,0.00036551105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013422308,0.0005042002,0.0005397713,0.0022495666,0.0005774532,0.0029891718,0.0004834731,0.0006026335,0.007251019],"category_scores_gemma":[0.024983246,0.00022317473,0.00039163313,0.0027270839,0.0012114513,0.0033176455,0.0013845005,0.0007026229,0.0006472934],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010718977,0.00025002364,0.22603758,0.00086596055,0.0007361903,0.0008418405,0.0022810453,0.07319207,0.018465167,0.22352381,0.009146672,0.44358775],"study_design_scores_gemma":[0.000042343414,0.00036545686,0.23455444,0.00018081164,0.00018573568,0.0013291824,0.0012596237,0.1318747,0.011734475,0.59934473,0.018890712,0.00023777058],"about_ca_topic_score_codex":0.0013405605,"about_ca_topic_score_gemma":0.00089049264,"teacher_disagreement_score":0.007251019,"about_ca_system_score_codex":0.00073054584,"about_ca_system_score_gemma":0.00042679088,"threshold_uncertainty_score":0.024257064},"labels":[],"label_agreement":null},{"id":"W4407505400","doi":"10.1037/xlm0001438","title":"Tracking the dynamic word-by-word incremental reading through multimeasures.","year":2025,"lang":"en","type":"article","venue":"Journal of Experimental Psychology Learning Memory and Cognition","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Word (group theory); Psychology; Reading (process); Word recognition; Linguistics; Word length; Cognitive psychology; Word lists by frequency; Natural language processing; Computer science; Sentence","score_opus":0.02199783729644601,"score_gpt":0.36202037051378916,"score_spread":0.34002253321734316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407505400","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89804244,0.0006329261,0.08348607,0.00026799954,0.000105638996,0.00028938003,0.0006492598,0.0009003267,0.015625846],"genre_scores_gemma":[0.95096695,0.00027725947,0.045525406,0.00010542638,0.000024827425,0.00022019567,0.000340895,0.00014179407,0.0023971593],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99948287,0.000102960585,0.000028278031,0.00024498926,0.00011831348,0.000022583903],"domain_scores_gemma":[0.9951774,0.002852095,0.00089391955,0.0007746971,0.00018629433,0.00011568106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009724272,0.00029234673,0.00022946557,0.00048506196,0.00017685928,0.0010642796,0.0005677509,0.0004454296,0.0054452317],"category_scores_gemma":[0.012107228,0.0003218229,0.00016844383,0.00042104194,0.000671632,0.003058734,0.0010217158,0.00070807745,0.0008398092],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011554686,0.000613261,0.032791037,0.00097412453,0.00016339854,0.0003788787,0.0044538276,0.0021042775,0.4598851,0.018616395,0.0027973154,0.4760669],"study_design_scores_gemma":[0.00013549742,0.003582919,0.7582329,0.00018218327,0.00025999988,0.0016672588,0.0016310762,0.043698516,0.09397433,0.08413452,0.012325052,0.00017576826],"about_ca_topic_score_codex":0.0006661894,"about_ca_topic_score_gemma":0.0012610045,"teacher_disagreement_score":0.0054452317,"about_ca_system_score_codex":0.00025827417,"about_ca_system_score_gemma":0.00025846512,"threshold_uncertainty_score":0.018216133},"labels":[],"label_agreement":null},{"id":"W4407733387","doi":"10.1007/s10508-025-03105-6","title":"A Content Analysis of Lay Definitions of Romantic Chemistry","year":2025,"lang":"en","type":"article","venue":"Archives of Sexual Behavior","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"York University","keywords":"Content analysis; Romance; Psychology; Public health; Content (measure theory); Sexual behavior; Chemistry; Social psychology; Psychoanalysis; Social science; Sociology; Medicine","score_opus":0.05534291665120807,"score_gpt":0.31381662861688153,"score_spread":0.2584737119656735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407733387","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7686328,0.0019467821,0.11363385,0.004238937,0.00037494994,0.001884106,0.0139167,0.00060099794,0.094770886],"genre_scores_gemma":[0.92558944,0.00072818546,0.060962386,0.0002630063,0.000121536286,0.0013242955,0.0053374353,0.00032831953,0.0053453897],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.99330914,0.0042050206,0.0006228801,0.00039568622,0.0012941997,0.0001731141],"domain_scores_gemma":[0.9336277,0.05405226,0.002814987,0.0024192922,0.006636543,0.00044911564],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004705013,0.00030015872,0.00032685307,0.010403954,0.0017438135,0.0030616187,0.0005909666,0.00041815703,0.0058008595],"category_scores_gemma":[0.034587912,0.00018878216,0.00026865158,0.00931284,0.0026412962,0.0032503093,0.0021259815,0.0011696595,0.00053478923],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006687348,0.0002631388,0.04393888,0.002523165,0.00006673004,0.0005357087,0.27431974,0.00089074206,0.015115937,0.33360493,0.023961129,0.3041111],"study_design_scores_gemma":[0.00010502087,0.00030808678,0.13910115,0.0036745225,0.00022313102,0.0016037406,0.2843138,0.02378081,0.023183936,0.11831061,0.40522265,0.00017247214],"about_ca_topic_score_codex":0.0015744353,"about_ca_topic_score_gemma":0.0016413907,"teacher_disagreement_score":0.010403954,"about_ca_system_score_codex":0.0026352108,"about_ca_system_score_gemma":0.002369086,"threshold_uncertainty_score":0.024882793},"labels":[],"label_agreement":null},{"id":"W4408173301","doi":"10.26434/chemrxiv-2025-8z6h2","title":"MERMaid: Universal multimodal mining of chemical reactions from PDFs using vision-language models","year":2025,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; Canadian Institute for Advanced Research; University of Toronto","funders":"Canada First Research Excellence Fund; Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence","score_opus":0.02439863093086714,"score_gpt":0.3166167865368164,"score_spread":0.29221815560594927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408173301","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010474936,0.0007128593,0.8903714,0.00034626116,0.00016337464,0.00023719511,0.006601264,0.08770954,0.0033831524],"genre_scores_gemma":[0.1023001,0.00052281405,0.8768252,0.00042598258,0.00006354368,0.0005310808,0.012561741,0.0023760963,0.004393488],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991646,0.00013851242,0.00004984531,0.00033975553,0.00024987338,0.000057491878],"domain_scores_gemma":[0.99858034,0.00072435895,0.00011665436,0.00028447362,0.0002241316,0.00007000179],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014103579,0.0016609628,0.0007952857,0.0026886095,0.0007381454,0.0026162884,0.002412057,0.0015790039,0.00816145],"category_scores_gemma":[0.006457157,0.00069254264,0.0015562018,0.001280237,0.00063647475,0.0032656607,0.003476038,0.0017986472,0.00482051],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008297698,0.00029402762,0.0028250176,0.001488351,0.00030637666,0.00064172316,0.00086306024,0.030202135,0.062550925,0.019691503,0.07583051,0.80447656],"study_design_scores_gemma":[0.00009754256,0.0001394876,0.0011085988,0.00012345171,0.00007988845,0.00044709138,0.0003068536,0.80863166,0.0719439,0.05685798,0.0601542,0.000109299064],"about_ca_topic_score_codex":0.0035609116,"about_ca_topic_score_gemma":0.0070037763,"teacher_disagreement_score":0.00816145,"about_ca_system_score_codex":0.00089043035,"about_ca_system_score_gemma":0.001495901,"threshold_uncertainty_score":0.027302802},"labels":[],"label_agreement":null},{"id":"W4409163690","doi":"10.1007/s00766-025-00436-7","title":"Tracing content requirements in financial documents using multi-granularity text analysis","year":2025,"lang":"en","type":"article","venue":"Requirements Engineering","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg","keywords":"Tracing; Granularity; Computer science; Requirements analysis; Information retrieval; Software engineering; Programming language; Software","score_opus":0.05193979241721546,"score_gpt":0.3293895904679666,"score_spread":0.27744979805075115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409163690","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4459585,0.0017732662,0.5254078,0.000588589,0.00010898947,0.00088988605,0.008517655,0.013276953,0.0034784274],"genre_scores_gemma":[0.51321423,0.00035442333,0.47619605,0.00010223686,0.00008795131,0.0002686999,0.0083001545,0.00027483434,0.0012014568],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99764204,0.00044870106,0.00033069495,0.0005332529,0.00092936854,0.00011587525],"domain_scores_gemma":[0.9860074,0.0075190957,0.002868093,0.00090931926,0.0023310105,0.0003650282],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013104778,0.0011858913,0.0007931934,0.011529471,0.0007044469,0.00187924,0.00094007415,0.0010624223,0.0010842485],"category_scores_gemma":[0.009626158,0.00032172352,0.0006796052,0.004431171,0.00041335513,0.0020711715,0.00089264876,0.0008281896,0.000934017],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008405075,0.0005636866,0.04755989,0.0016972226,0.00019214387,0.0022665616,0.0021501114,0.03164845,0.11444213,0.0029687672,0.009528331,0.78614223],"study_design_scores_gemma":[0.00007627328,0.000357799,0.04591497,0.00017143562,0.00018932071,0.0011490072,0.0012372984,0.84200346,0.09030315,0.0063890168,0.01206828,0.00013996866],"about_ca_topic_score_codex":0.0043047415,"about_ca_topic_score_gemma":0.0051022116,"teacher_disagreement_score":0.011529471,"about_ca_system_score_codex":0.0008805509,"about_ca_system_score_gemma":0.00096917787,"threshold_uncertainty_score":0.008559346},"labels":[],"label_agreement":null},{"id":"W4409313533","doi":"10.3390/app15084134","title":"LLM-Enhanced Framework for Building Domain-Specific Lexicon for Urban Power Grid Design","year":2025,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"State Grid Jiangsu Electric Power","keywords":"Computer science; Power grid; Lexicon; Architectural engineering; Power (physics); Artificial intelligence; Engineering; Physics","score_opus":0.028962359607354066,"score_gpt":0.3241110746509073,"score_spread":0.2951487150435533,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409313533","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032628807,0.00028034128,0.9781081,0.00022484995,0.000054500575,0.0002722826,0.0019944734,0.011878987,0.003923638],"genre_scores_gemma":[0.053452972,0.00036905575,0.9321036,0.00020001516,0.00004120467,0.00061314355,0.009042092,0.00079339027,0.0033844763],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911493,0.0002302598,0.0001567744,0.00019828313,0.00024255243,0.000057292422],"domain_scores_gemma":[0.99937147,0.00020418031,0.000073196774,0.00013776126,0.00017703783,0.000036454225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008859302,0.001087085,0.00070523453,0.0031976935,0.00064045645,0.002428272,0.0012950808,0.0007669689,0.008027661],"category_scores_gemma":[0.0033881953,0.0006523298,0.0022668992,0.0019181591,0.0004956727,0.002451979,0.0024324202,0.0012152554,0.0053541516],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023068812,0.00023395855,0.0041072513,0.0021133823,0.00022834522,0.0020700465,0.0022493158,0.07524488,0.03406478,0.14095807,0.054180462,0.68431884],"study_design_scores_gemma":[0.00007062644,0.00009493129,0.0012982643,0.000286431,0.00013762324,0.0013881689,0.00074569933,0.66009825,0.012559289,0.0867677,0.2364522,0.000100791265],"about_ca_topic_score_codex":0.0055793775,"about_ca_topic_score_gemma":0.011379961,"teacher_disagreement_score":0.008027661,"about_ca_system_score_codex":0.0009952015,"about_ca_system_score_gemma":0.0022606181,"threshold_uncertainty_score":0.02685523},"labels":[],"label_agreement":null},{"id":"W4409603307","doi":"10.61091/jcmcc127b-142","title":"Research on the Optimization Method of Japanese Text Information Dissemination Path Based on Graph Theory","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"China Scholarship Council","keywords":"Computer science; Path (computing); Graph theory; Graph; Theoretical computer science; Information retrieval; Mathematics; Combinatorics; Computer network","score_opus":0.015817120619428676,"score_gpt":0.3401000032793355,"score_spread":0.3242828826599068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409603307","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039023105,0.0011461568,0.9497314,0.0007742286,0.00007628554,0.00014142644,0.00015855815,0.00021418062,0.008734703],"genre_scores_gemma":[0.793767,0.003514878,0.19050886,0.00022148022,0.0001043353,0.00049019966,0.00052487,0.00020403406,0.010664363],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989297,0.0004667462,0.000042048297,0.00023538498,0.00020963326,0.00011652858],"domain_scores_gemma":[0.9970294,0.002003349,0.00027246517,0.00013068433,0.0004477497,0.00011641766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015776715,0.00096909015,0.0008471911,0.0017949451,0.00080276985,0.0015196973,0.0013029992,0.0009596513,0.0034756304],"category_scores_gemma":[0.007157026,0.00046474877,0.0009483069,0.001715912,0.00076726277,0.003922013,0.0009598584,0.0011357605,0.00033766107],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008207137,0.000070103495,0.002655961,0.00038628324,0.00008362068,0.000120977544,0.00035480355,0.8023216,0.002367082,0.11152643,0.0042927423,0.07573837],"study_design_scores_gemma":[0.000012346136,0.000038572278,0.00038806684,0.000015739499,0.000030875337,0.00004093014,0.000086223095,0.97552013,0.00046880427,0.021470372,0.0019142013,0.000013635425],"about_ca_topic_score_codex":0.009943641,"about_ca_topic_score_gemma":0.0054334356,"teacher_disagreement_score":0.009943641,"about_ca_system_score_codex":0.002366626,"about_ca_system_score_gemma":0.0024306457,"threshold_uncertainty_score":0.019771516},"labels":[],"label_agreement":null},{"id":"W4409603675","doi":"10.61091/jcmcc127b-196","title":"A Study on Improving Semantic Consistency of Translation Systems by Combining Dynamic Computing Methods in English Corpus","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Jilin Office of Philosophy and Social Science; Henan Office of Philosophy and Social Science","keywords":"Computer science; Consistency (knowledge bases); Natural language processing; Translation (biology); Artificial intelligence; Information retrieval; Chemistry","score_opus":0.016288306506166188,"score_gpt":0.32592750652418884,"score_spread":0.30963920001802264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409603675","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23228136,0.0027269912,0.7559364,0.00046887778,0.00014548127,0.00012384502,0.00009062268,0.0015522975,0.0066740704],"genre_scores_gemma":[0.75790554,0.0013411178,0.23664667,0.00015263137,0.00010588985,0.000109066066,0.00036961128,0.00028840642,0.0030810046],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99830014,0.00061457243,0.00014370639,0.00046915712,0.00038484164,0.00008753585],"domain_scores_gemma":[0.9977233,0.001014441,0.00013113796,0.00043330685,0.0006487743,0.00004897572],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030579474,0.00071571826,0.00079201406,0.0010380732,0.00080080325,0.0011185596,0.0008645316,0.00057233707,0.0012977368],"category_scores_gemma":[0.008366267,0.00043383884,0.00081802753,0.0016484585,0.00072396663,0.003993249,0.0010079345,0.0008811258,0.00028821928],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006915632,0.00028881256,0.010514762,0.00055269623,0.00036150787,0.00023222515,0.0011941809,0.11270499,0.064525515,0.029137705,0.0021333187,0.7776627],"study_design_scores_gemma":[0.00008055296,0.00059141143,0.00637534,0.00005004715,0.00025408188,0.00037071275,0.00038028325,0.9203853,0.054100875,0.0075089345,0.009837061,0.00006539928],"about_ca_topic_score_codex":0.005948469,"about_ca_topic_score_gemma":0.0031762784,"teacher_disagreement_score":0.005948469,"about_ca_system_score_codex":0.000926613,"about_ca_system_score_gemma":0.0013837274,"threshold_uncertainty_score":0.016172111},"labels":[],"label_agreement":null},{"id":"W4409603800","doi":"10.61091/jcmcc127b-230","title":"Value Assessment and Linguistic Feature Mining Based on Regression Analysis in Ancient Literary Texts","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Value (mathematics); Feature (linguistics); Linguistics; Regression; Natural language processing; Regression analysis; Artificial intelligence; History; Computer science; Statistics; Mathematics; Philosophy","score_opus":0.008985441968079889,"score_gpt":0.3100319215638725,"score_spread":0.3010464795957926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409603800","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.613182,0.0014461318,0.37064368,0.0008110885,0.000093041344,0.00025690455,0.001307409,0.0009125894,0.011347058],"genre_scores_gemma":[0.92213535,0.0003640893,0.07446976,0.000036072848,0.0000554765,0.0001888857,0.0010025603,0.000063275904,0.0016845235],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9956138,0.0018483503,0.00040717275,0.0007355557,0.0012074317,0.00018769706],"domain_scores_gemma":[0.9869039,0.008771009,0.0014052404,0.0004958705,0.0022589092,0.00016507159],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034399081,0.0008703856,0.0007333052,0.009816211,0.0007297794,0.0020392071,0.00084180216,0.00049091375,0.001515694],"category_scores_gemma":[0.025519108,0.00024662827,0.0009952731,0.0073094736,0.0008119019,0.0027370346,0.0009897801,0.00081780186,0.00056470616],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005098996,0.00034315078,0.14488874,0.0010443394,0.0005032305,0.0011786313,0.0036343578,0.044432547,0.009932016,0.02116768,0.00469106,0.76767427],"study_design_scores_gemma":[0.000046326688,0.00029011726,0.12561882,0.00024498004,0.00032418178,0.0006823503,0.0030707184,0.8075598,0.0125613,0.040654328,0.008783781,0.00016333687],"about_ca_topic_score_codex":0.002488952,"about_ca_topic_score_gemma":0.0019738192,"teacher_disagreement_score":0.009816211,"about_ca_system_score_codex":0.0010767628,"about_ca_system_score_gemma":0.00068009854,"threshold_uncertainty_score":0.018192172},"labels":[],"label_agreement":null},{"id":"W4409813933","doi":"10.1016/j.procs.2025.03.116","title":"Harnessing Large Language Models for Precision Topic Extraction and Technology Patent Nomination: A GPT-centric Methodology","year":2025,"lang":"en","type":"article","venue":"Procedia Computer Science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Cégep de Rimouski; Université du Québec à Rimouski","funders":"","keywords":"Computer science; Nomination; Patent analysis; Extraction (chemistry); Data science; Data mining","score_opus":0.04369533804685807,"score_gpt":0.3513519287723578,"score_spread":0.3076565907254998,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409813933","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019512601,0.0008461376,0.9744221,0.00092592544,0.0000819682,0.00019275548,0.0009293712,0.0013662202,0.0017229103],"genre_scores_gemma":[0.46305147,0.0015318893,0.52278745,0.00054076815,0.00064953975,0.0010187819,0.005334745,0.00043688016,0.0046484414],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976399,0.0011662856,0.0001473533,0.00057191553,0.0003675486,0.00010700253],"domain_scores_gemma":[0.9876809,0.009922191,0.00080072175,0.0006981569,0.0007474538,0.00015060331],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034737498,0.0013232362,0.0009218327,0.0039832573,0.0008704234,0.0027978534,0.001241963,0.0013095451,0.0019131701],"category_scores_gemma":[0.017463025,0.00050864473,0.0019096677,0.0034413897,0.0007572367,0.0041606,0.0018054586,0.0022155526,0.0021229954],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004068278,0.00043002417,0.012071333,0.0010158602,0.0005602521,0.0007024991,0.0024491278,0.15131003,0.020057827,0.06597355,0.014941331,0.73008144],"study_design_scores_gemma":[0.000035263307,0.000081110054,0.001547195,0.00006244336,0.0001105443,0.00019597857,0.00023454175,0.93119174,0.004232899,0.053184006,0.009075004,0.000049372273],"about_ca_topic_score_codex":0.003541529,"about_ca_topic_score_gemma":0.0053707287,"teacher_disagreement_score":0.0039832573,"about_ca_system_score_codex":0.0010514681,"about_ca_system_score_gemma":0.002148942,"threshold_uncertainty_score":0.018371105},"labels":[],"label_agreement":null},{"id":"W4409973579","doi":"10.1145/3698204.3716444","title":"Integrating Eye Tracking, Feature Use, and Emotional Valence: A Multimodal Approach to Evaluating Search Interfaces","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"Natural Sciences and Engineering Research Council of Canada; University of Regina","keywords":"Computer science; Eye tracking; Emotional valence; Feature (linguistics); Artificial intelligence; Valence (chemistry); Feature extraction; Computer vision; Human–computer interaction; Psychology; Cognition","score_opus":0.04454853921092032,"score_gpt":0.38224133973688423,"score_spread":0.3376928005259639,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409973579","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92629826,0.00035899522,0.06568071,0.00009041339,0.000039543123,0.0009827361,0.00057493575,0.0004310318,0.005543406],"genre_scores_gemma":[0.9507792,0.0001875019,0.046161573,0.00009720274,0.000035414992,0.0011949297,0.0002824153,0.00005952036,0.0012021794],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9980404,0.000817161,0.00016841118,0.00023711649,0.0006238852,0.00011304402],"domain_scores_gemma":[0.9927167,0.0040750937,0.0011722213,0.00039273675,0.0014007299,0.00024267053],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026942613,0.00095006987,0.0005702967,0.002620279,0.0003458778,0.0014038613,0.00038830756,0.00070331834,0.0018878145],"category_scores_gemma":[0.012952475,0.00029020567,0.00046826713,0.0011198828,0.0003898565,0.0013100409,0.001185858,0.0005674755,0.0002810007],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0048890384,0.0022858186,0.13744359,0.0023176481,0.00052044337,0.00026864646,0.0066041187,0.0038612422,0.5311493,0.0013462557,0.0016041681,0.30770963],"study_design_scores_gemma":[0.00028648012,0.007284752,0.8674742,0.00026878872,0.000577222,0.00056698354,0.0034903856,0.039898764,0.07548925,0.002444367,0.0019337098,0.00028502572],"about_ca_topic_score_codex":0.0009678441,"about_ca_topic_score_gemma":0.0016833616,"teacher_disagreement_score":0.0026942613,"about_ca_system_score_codex":0.00044418048,"about_ca_system_score_gemma":0.0002509266,"threshold_uncertainty_score":0.014248788},"labels":[],"label_agreement":null},{"id":"W4409973737","doi":"10.1145/3698204.3716481","title":"NeuroPhysIIR: International Workshop on NeuroPhysiological Approaches for Interactive Information Retrieval","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"University of Toronto; RMIT University","keywords":"Computer science; Neurophysiology; Information retrieval; Neuroscience; Psychology","score_opus":0.03177285531587556,"score_gpt":0.30303675902391414,"score_spread":0.2712639037080386,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409973737","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0061470205,0.06331032,0.8455178,0.017113727,0.010582229,0.0013453236,0.0037960003,0.012541948,0.03964559],"genre_scores_gemma":[0.07222053,0.0391147,0.74260336,0.0070547573,0.0060528875,0.0026107708,0.010908457,0.0045780945,0.11485644],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.993952,0.002273117,0.0005482351,0.0010842588,0.001720571,0.00042179247],"domain_scores_gemma":[0.9908297,0.0048997165,0.00024091089,0.0012063744,0.0019254251,0.00089788646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012038287,0.0020386425,0.0021841677,0.0035823446,0.0016053085,0.00903799,0.004021204,0.0035866485,0.03914408],"category_scores_gemma":[0.015191225,0.0010247115,0.0023772097,0.0029459344,0.0023053237,0.010139042,0.004811376,0.005863932,0.017426882],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006770301,0.00032424802,0.0004575269,0.0017906326,0.00023387583,0.00044508616,0.001640623,0.00203031,0.020924967,0.046460435,0.39666307,0.52835214],"study_design_scores_gemma":[0.0001974748,0.00049966195,0.0028386286,0.0013471673,0.00021203376,0.0014383767,0.001021651,0.031929363,0.015964177,0.11156495,0.8327447,0.0002417775],"about_ca_topic_score_codex":0.0056035016,"about_ca_topic_score_gemma":0.004945939,"teacher_disagreement_score":0.03914408,"about_ca_system_score_codex":0.0020912616,"about_ca_system_score_gemma":0.0038970786,"threshold_uncertainty_score":0.13094997},"labels":[],"label_agreement":null},{"id":"W4410016295","doi":"10.21917/ijsc.2025.0526","title":"A NATURAL LANGUAGE PROCESSING APPROACH TO COMPARATIVE SENTIMENT AND TOPIC ANALYSIS OF ENGLISH NATIONAL ANTHEMS","year":2025,"lang":"en","type":"article","venue":"ICTACT Journal on Soft Computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Natural language processing; Computer science; Linguistics; Sentiment analysis; Natural (archaeology); Artificial intelligence; History; Archaeology","score_opus":0.01574438382345438,"score_gpt":0.3384823788672222,"score_spread":0.3227379950437678,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410016295","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43654978,0.0014349122,0.50869954,0.001562466,0.00030047135,0.0020241998,0.008300994,0.0018500304,0.039277583],"genre_scores_gemma":[0.62387806,0.0006611238,0.35891414,0.00014227637,0.00020138604,0.0019952988,0.0055860835,0.00023886806,0.008382727],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99867576,0.0006030256,0.00012959022,0.0002736516,0.0002442963,0.0000736175],"domain_scores_gemma":[0.9968382,0.0021804736,0.00027673205,0.00012609773,0.00052019407,0.000058148482],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002074548,0.0004253166,0.00034221724,0.003497067,0.00097695,0.0023428614,0.00053457136,0.00032617914,0.0034191262],"category_scores_gemma":[0.005947674,0.00020217185,0.0005587536,0.002639695,0.00072869734,0.0015598276,0.0007704491,0.0007075065,0.00066016364],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070205785,0.00034244783,0.014617366,0.0021617492,0.00016391464,0.0025774697,0.0777335,0.0052924436,0.10178523,0.057615727,0.018565489,0.7184426],"study_design_scores_gemma":[0.00018314246,0.0006355869,0.19616777,0.00069917267,0.00030688802,0.0024414335,0.10154694,0.20801495,0.05517354,0.05875409,0.37576228,0.0003143086],"about_ca_topic_score_codex":0.0029553755,"about_ca_topic_score_gemma":0.003975656,"teacher_disagreement_score":0.003497067,"about_ca_system_score_codex":0.0012365745,"about_ca_system_score_gemma":0.00080417906,"threshold_uncertainty_score":0.011438131},"labels":[],"label_agreement":null},{"id":"W4410087324","doi":"10.1109/wi-iat62293.2024.00083","title":"Using LLMs to Analyze Antecedent, Behavior and Consequence Narrative Recordings in Behavioral Health Science","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Antecedent (behavioral psychology); Narrative; Behavioural sciences; Psychology; Computer science; Cognitive psychology; Social psychology; Linguistics; Psychotherapist","score_opus":0.06715577909876917,"score_gpt":0.43577917777842723,"score_spread":0.36862339867965804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410087324","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09891992,0.0008663043,0.8862459,0.0019979274,0.00012502789,0.00085786835,0.004191038,0.0031075857,0.003688446],"genre_scores_gemma":[0.48888612,0.0004388851,0.5023632,0.00057056476,0.00009770704,0.0015190495,0.0042308136,0.00016007129,0.0017335048],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995379,0.0030355616,0.0003244694,0.0007598148,0.00038950515,0.00011168847],"domain_scores_gemma":[0.96333605,0.030788833,0.0025611327,0.0015801917,0.0014455875,0.00028815187],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008919638,0.0011603516,0.00055167987,0.003330549,0.00070950476,0.0019139671,0.0012139555,0.0009596957,0.003801983],"category_scores_gemma":[0.03560203,0.00042100763,0.0014879502,0.0020444307,0.0009575858,0.0022228092,0.0017699022,0.001899428,0.0013285733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000932569,0.0010342695,0.17606197,0.0026479557,0.00061065535,0.0010457832,0.012673815,0.092809394,0.015640546,0.0333137,0.010728395,0.65250105],"study_design_scores_gemma":[0.00006959916,0.00038074524,0.0392358,0.00071925437,0.00018424935,0.0002557773,0.0040765833,0.82464314,0.008853834,0.10419123,0.017250039,0.0001397832],"about_ca_topic_score_codex":0.009479827,"about_ca_topic_score_gemma":0.018746106,"teacher_disagreement_score":0.009479827,"about_ca_system_score_codex":0.0018257317,"about_ca_system_score_gemma":0.0027524945,"threshold_uncertainty_score":0.04717213},"labels":[],"label_agreement":null},{"id":"W4410140237","doi":"10.1007/s41870-025-02541-w","title":"TextAI 3.0 (Multimodal): multimodal sentiment analysis using attention-enabled ensemble-based deep learning in hyperbolic space","year":2025,"lang":"en","type":"article","venue":"International Journal of Information Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Artificial intelligence; Space (punctuation); Sentiment analysis; Ensemble learning; Deep learning; Hyperbolic space; Machine learning; Mathematics","score_opus":0.005213480878788831,"score_gpt":0.281656305663714,"score_spread":0.2764428247849252,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410140237","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06934915,0.0012684603,0.7103062,0.00077814475,0.00070874067,0.00052395347,0.022634856,0.18240814,0.012022376],"genre_scores_gemma":[0.33397886,0.0007543614,0.5848984,0.0005846538,0.0002501245,0.0011499453,0.039432496,0.009671793,0.029279478],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999757,0.000050593073,0.000013814951,0.00006389193,0.000076339806,0.000038335944],"domain_scores_gemma":[0.999706,0.0000928136,0.000020586875,0.00005047454,0.000100713725,0.000029411734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007114795,0.001518691,0.0006541643,0.0010328036,0.0003485006,0.0009983063,0.0012055706,0.000802548,0.01920851],"category_scores_gemma":[0.0020015046,0.0003889155,0.00080696243,0.00077347335,0.0001799765,0.0015264193,0.0016357009,0.0016124232,0.006967764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010172512,0.00067262235,0.0035459704,0.00074402033,0.00057064305,0.00029207088,0.00048702926,0.046476785,0.048678096,0.0065472326,0.20159633,0.6893719],"study_design_scores_gemma":[0.000107948596,0.0001928785,0.0016256843,0.00003939803,0.000072327784,0.00006638848,0.00009300898,0.9440163,0.021109294,0.008531127,0.024078758,0.00006693259],"about_ca_topic_score_codex":0.0046669985,"about_ca_topic_score_gemma":0.009219391,"teacher_disagreement_score":0.01920851,"about_ca_system_score_codex":0.00042101264,"about_ca_system_score_gemma":0.00061606494,"threshold_uncertainty_score":0.06425887},"labels":[],"label_agreement":null},{"id":"W4410355141","doi":"10.36315/2025inpact142","title":"AN ECOLOGICAL APPROACH TO THEORY-OF-MIND MEASUREMENT: CREATION OF THE EV-TOMI FROM OPEN-ENDED REPORTS","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Ecology; Biology","score_opus":0.04203306755129314,"score_gpt":0.3174950439469187,"score_spread":0.27546197639562553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410355141","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3323178,0.00046356907,0.59808165,0.0015427344,0.00040625664,0.019750861,0.002801392,0.001320447,0.043315247],"genre_scores_gemma":[0.38944212,0.0002886402,0.56891197,0.0005170957,0.00007890743,0.036801558,0.0022213713,0.000281017,0.001457334],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.93423146,0.041364193,0.0077449903,0.003242552,0.012345804,0.001071099],"domain_scores_gemma":[0.8160138,0.098762974,0.017095407,0.016533323,0.049278893,0.0023157105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.09928137,0.0013162249,0.0008906323,0.010823808,0.0023837017,0.0046610194,0.0020863325,0.0009594218,0.0021038635],"category_scores_gemma":[0.19294779,0.0012305381,0.0019226493,0.006564819,0.004418378,0.00437787,0.008572458,0.0031743376,0.00077920005],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004745537,0.001342814,0.3176689,0.0027581302,0.0005786761,0.00038311892,0.1284674,0.0035834573,0.011661436,0.08691764,0.008935845,0.43722802],"study_design_scores_gemma":[0.0002502837,0.0016917995,0.6260221,0.0022023988,0.0002911814,0.0011217064,0.06363781,0.029654881,0.015916253,0.15352969,0.10505642,0.0006254688],"about_ca_topic_score_codex":0.0035699578,"about_ca_topic_score_gemma":0.006212837,"teacher_disagreement_score":0.09928137,"about_ca_system_score_codex":0.0030069249,"about_ca_system_score_gemma":0.005486058,"threshold_uncertainty_score":0.52505636},"labels":[],"label_agreement":null},{"id":"W4410436552","doi":"10.1142/s1793351x25440015","title":"Translative Research Assistant: A Retrieval-Augmented Generation Pipeline Refinement with Keyword Extraction Using Extended Scalable Betweenness Centrality","year":2025,"lang":"en","type":"article","venue":"International Journal of Semantic Computing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Northern British Columbia","funders":"","keywords":"Betweenness centrality; Computer science; Pipeline (software); Scalability; Information retrieval; Centrality; Keyword extraction; Keyword search; Artificial intelligence; Data mining; Database","score_opus":0.07189166716247476,"score_gpt":0.42396534476847936,"score_spread":0.3520736776060046,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410436552","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015247892,0.00027964398,0.96332663,0.00060479273,0.00012981133,0.0006889868,0.000666909,0.016086768,0.0029686256],"genre_scores_gemma":[0.11598614,0.00017202205,0.8742648,0.00024972312,0.000102755694,0.00064448215,0.0021602965,0.0012844192,0.0051353676],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99336606,0.003151296,0.0004759953,0.0013135371,0.0014839899,0.00020906575],"domain_scores_gemma":[0.9829868,0.008457114,0.0009876046,0.0033724848,0.0037040822,0.0004918914],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056156386,0.0015092756,0.0010208762,0.0040316447,0.0011528553,0.0029768804,0.0020874597,0.0014109719,0.008511818],"category_scores_gemma":[0.03092735,0.00052395347,0.0010625942,0.002571679,0.0009590618,0.004662331,0.003682819,0.0014085317,0.006113503],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088225596,0.00040304693,0.0023777718,0.0012671476,0.00012789489,0.0009622921,0.0040632254,0.010100255,0.06444348,0.019324867,0.019051868,0.8769959],"study_design_scores_gemma":[0.00058922713,0.0010303786,0.0034814617,0.0002951234,0.00039726528,0.0023516535,0.0042516035,0.63829124,0.14801423,0.07812476,0.12283693,0.00033610006],"about_ca_topic_score_codex":0.0019262395,"about_ca_topic_score_gemma":0.0026285858,"teacher_disagreement_score":0.008511818,"about_ca_system_score_codex":0.000948071,"about_ca_system_score_gemma":0.0026506658,"threshold_uncertainty_score":0.02969867},"labels":[],"label_agreement":null},{"id":"W4410512048","doi":"10.33423/jabe.v27i3.7646","title":"New Recommendation Agent to Identify Innovators Utilizing User-Generated Content","year":2025,"lang":"en","type":"article","venue":"Journal of Applied Business and Economics","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Content (measure theory); Information retrieval; World Wide Web; Internet privacy; Mathematics","score_opus":0.03911475642310851,"score_gpt":0.293659834904992,"score_spread":0.25454507848188346,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410512048","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1022415,0.0010203193,0.874805,0.0015993778,0.00036801148,0.0008854156,0.0020041121,0.0075865244,0.009489787],"genre_scores_gemma":[0.35243592,0.00033613865,0.62906516,0.00044114742,0.0001754175,0.0005363145,0.0022558805,0.00010403963,0.014650002],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99892324,0.00022848202,0.00010969078,0.0003875414,0.00028245614,0.00006857802],"domain_scores_gemma":[0.99652654,0.0010325819,0.0003775369,0.00051606126,0.0013177145,0.0002296107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017319995,0.0008197395,0.0013964762,0.002150194,0.00077732967,0.0013567113,0.0016972414,0.0018459528,0.0026527038],"category_scores_gemma":[0.005532958,0.00030496664,0.00065707695,0.0013637858,0.00027647323,0.0018583686,0.00078943634,0.0011408183,0.001690899],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009135543,0.0016136266,0.055822358,0.00034667764,0.0006226014,0.0003518926,0.00038129234,0.06357062,0.015997438,0.014712002,0.030654335,0.8150136],"study_design_scores_gemma":[0.0000803782,0.00017474,0.0025285531,0.000019462474,0.00009758179,0.00016555555,0.000052112104,0.98338807,0.0045268596,0.00196249,0.0069646016,0.000039598057],"about_ca_topic_score_codex":0.009522058,"about_ca_topic_score_gemma":0.0150609445,"teacher_disagreement_score":0.009522058,"about_ca_system_score_codex":0.0007630639,"about_ca_system_score_gemma":0.0012530974,"threshold_uncertainty_score":0.018933237},"labels":[],"label_agreement":null},{"id":"W4410637109","doi":"10.1145/3701716.3715869","title":"Query Understanding in LLM-based Conversational Information Seeking","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Singapore Management University; Universiteit van Amsterdam; Institute for Catastrophic Loss Reduction","keywords":"Computer science; Information retrieval; World Wide Web; Natural language processing","score_opus":0.01799660445673285,"score_gpt":0.27139216909930397,"score_spread":0.2533955646425711,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410637109","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05767658,0.0017973458,0.9271775,0.003028443,0.00008595105,0.00044530025,0.0003042638,0.0022311967,0.0072533195],"genre_scores_gemma":[0.6416574,0.000799179,0.35166034,0.0006601887,0.000109377994,0.00045622562,0.0005673409,0.0003178545,0.0037720366],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9859169,0.009434309,0.00066619704,0.0013540462,0.002097773,0.0005307325],"domain_scores_gemma":[0.9770445,0.017658368,0.0011670211,0.0014840921,0.0021850804,0.00046092639],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011046111,0.0009773606,0.0012109586,0.0021451795,0.0020008266,0.0054323226,0.0020007761,0.0023790163,0.0030603863],"category_scores_gemma":[0.051515304,0.0009206947,0.0011061656,0.0013925929,0.0022453738,0.009874138,0.004990164,0.0025566868,0.0010408042],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016664534,0.00041275987,0.010585478,0.0024026767,0.0003405186,0.0012216697,0.066170186,0.06909347,0.070467055,0.244376,0.016413335,0.5168504],"study_design_scores_gemma":[0.00005180922,0.0002451311,0.002478713,0.00022111961,0.00015081017,0.00059902563,0.0076323673,0.787177,0.017261006,0.16034837,0.023621103,0.00021341845],"about_ca_topic_score_codex":0.0069310917,"about_ca_topic_score_gemma":0.0049728937,"teacher_disagreement_score":0.011046111,"about_ca_system_score_codex":0.0029037045,"about_ca_system_score_gemma":0.0021922407,"threshold_uncertainty_score":0.058418095},"labels":[],"label_agreement":null},{"id":"W4411113273","doi":"10.18653/v1/2024.inlg-main.17","title":"Transfer-Learning based on Extract, Paraphrase and Compress Models for Neural Abstractive Multi-Document Summarization","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Paraphrase; Automatic summarization; Computer science; Natural language processing; Artificial intelligence; Transfer of learning; Information retrieval","score_opus":0.026775842732887505,"score_gpt":0.3087025949968748,"score_spread":0.2819267522639873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411113273","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.075361714,0.001273895,0.9157659,0.00060774706,0.00010685465,0.00016726048,0.00030446326,0.0039767916,0.002435333],"genre_scores_gemma":[0.74856144,0.00072034483,0.2370612,0.0003506896,0.00017281254,0.00046529592,0.0016086277,0.0002615872,0.010797921],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996687,0.00011927576,0.000027311438,0.00008474809,0.000058999518,0.000040967352],"domain_scores_gemma":[0.998855,0.0006756511,0.00008965622,0.00012106239,0.00021241783,0.00004617889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011768955,0.00096313155,0.00074640766,0.00066233124,0.00033933105,0.0006384091,0.0011725365,0.00091540924,0.0026596906],"category_scores_gemma":[0.0037573264,0.00032777767,0.0006719012,0.00070888223,0.00052024046,0.0018857958,0.00089019665,0.002180299,0.0010768895],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000211292,0.00030567354,0.00082652405,0.00019123343,0.00010458705,0.000119722485,0.00021365832,0.52274066,0.012828893,0.0052346196,0.0038968672,0.45332623],"study_design_scores_gemma":[0.000005765994,0.00007227382,0.00011849498,0.0000056875742,0.000011061501,0.000012323849,0.000015450067,0.99309736,0.0030569818,0.0031856757,0.00041364177,0.0000052428227],"about_ca_topic_score_codex":0.0035888827,"about_ca_topic_score_gemma":0.0060960306,"teacher_disagreement_score":0.0035888827,"about_ca_system_score_codex":0.0009486263,"about_ca_system_score_gemma":0.0007298715,"threshold_uncertainty_score":0.008897483},"labels":[],"label_agreement":null},{"id":"W4411236083","doi":"10.1145/3744644","title":"LLM-Cure: LLM-Based Competitor User Review Analysis for Feature Enhancement","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Queen's University; Université du Québec à Montréal","funders":"","keywords":"Computer science; Feature (linguistics); Artificial intelligence","score_opus":0.046042394086885305,"score_gpt":0.3511223933018579,"score_spread":0.3050799992149726,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411236083","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18613519,0.007776779,0.6033106,0.0019767147,0.00093055225,0.0037184746,0.027516574,0.1573546,0.011280518],"genre_scores_gemma":[0.39253986,0.0008433176,0.55871063,0.0009345908,0.00044435667,0.0014781277,0.032530155,0.002025878,0.010493102],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99326694,0.0021990328,0.00059084484,0.001392854,0.0023126993,0.0002375773],"domain_scores_gemma":[0.9838418,0.0070839706,0.0023809373,0.0014023995,0.004749075,0.00054189836],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047212476,0.002401,0.001333626,0.009213857,0.0006525779,0.0018282013,0.0018363126,0.001332357,0.0029368722],"category_scores_gemma":[0.02209772,0.0004958334,0.0018354154,0.0025316903,0.00039140193,0.002151292,0.0015920231,0.0012073922,0.004207655],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011922916,0.0010070722,0.044489413,0.003266103,0.0006790597,0.0011964041,0.001399245,0.012186871,0.04146761,0.0030556463,0.124237806,0.76582247],"study_design_scores_gemma":[0.00027250015,0.0009761006,0.02616002,0.00019091077,0.00037807252,0.0012839524,0.00064499635,0.8766328,0.03205218,0.0043252124,0.056808356,0.0002749398],"about_ca_topic_score_codex":0.006271653,"about_ca_topic_score_gemma":0.017006157,"teacher_disagreement_score":0.009213857,"about_ca_system_score_codex":0.0011658698,"about_ca_system_score_gemma":0.0022639355,"threshold_uncertainty_score":0.024968684},"labels":[],"label_agreement":null},{"id":"W4411259709","doi":"10.1101/2025.06.13.25329541","title":"Automation of Systematic Reviews with Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Public Health Ontario; University Health Network; Ottawa Hospital; University of Alberta; St. Michael's Hospital; University of British Columbia; Mount Sinai Hospital; University of Calgary; McGill University; University of Ottawa; Vector Institute; Wilfrid Laurier University; University of Waterloo; University of Toronto","funders":"","keywords":"Automation; Computer science; Systems engineering; Engineering; Mechanical engineering","score_opus":0.03430894453732251,"score_gpt":0.3234808861615266,"score_spread":0.28917194162420407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411259709","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012351866,0.0031603137,0.9286479,0.0032265675,0.00046909248,0.012708649,0.0049464502,0.030928234,0.0035609656],"genre_scores_gemma":[0.03429917,0.0007932429,0.94990873,0.0005724138,0.00012523352,0.010120745,0.0024592977,0.001298368,0.00042275147],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.57591313,0.3150223,0.0631388,0.017222183,0.027068134,0.0016355137],"domain_scores_gemma":[0.16529016,0.66558117,0.03305399,0.090499654,0.04416492,0.0014101281],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3451886,0.0041849953,0.005736387,0.015356283,0.0031051433,0.012927767,0.005899989,0.002260901,0.008552042],"category_scores_gemma":[0.69027215,0.0047692093,0.010286694,0.010833503,0.002289626,0.009202734,0.014728538,0.004320409,0.006057298],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002499025,0.00045622312,0.007842939,0.027697053,0.0039046188,0.0006317784,0.008930924,0.030907454,0.011956101,0.017313385,0.032821685,0.8550388],"study_design_scores_gemma":[0.008225611,0.002113437,0.01752048,0.024384324,0.00847602,0.0012619296,0.0048990697,0.43664876,0.052434824,0.21645086,0.22542568,0.0021589678],"about_ca_topic_score_codex":0.0063528996,"about_ca_topic_score_gemma":0.01244848,"teacher_disagreement_score":0.6548114,"about_ca_system_score_codex":0.0054668216,"about_ca_system_score_gemma":0.036917854,"threshold_uncertainty_score":0.80749905},"labels":[],"label_agreement":null},{"id":"W4411656759","doi":"10.51847/xz17igjvgz","title":"10.51847/xZ17iGjvGz","year":2000,"lang":"en","type":"article","venue":"Time to knit","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Cluster analysis; Computer science; Data mining; Artificial intelligence; Algorithm","score_opus":0.005534351542979675,"score_gpt":0.2057611130604097,"score_spread":0.20022676151743002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411656759","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062073036,0.0013343504,0.012079797,0.0014831284,0.0014660453,0.0003193435,0.002226121,0.0067528533,0.96813095],"genre_scores_gemma":[0.010764347,0.000663435,0.0062515344,0.00045227783,0.0001697032,0.00014287948,0.0027743012,0.0008681672,0.97791344],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99956876,0.000053449043,0.00003481533,0.00014056655,0.00012839635,0.000074029434],"domain_scores_gemma":[0.9992962,0.00019199323,0.000040283976,0.00015633237,0.00020880016,0.00010643498],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00068353885,0.0014836334,0.00085651217,0.0020270408,0.00090743863,0.0032869673,0.0011064374,0.003167403,0.91478395],"category_scores_gemma":[0.0011581728,0.00041777352,0.00058091193,0.0025525228,0.0007123146,0.0017729637,0.0015702606,0.0011363687,0.9232123],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040003157,0.00022846603,0.0016514993,0.0005198965,0.000046011286,0.00039557868,0.00014011741,0.0009948855,0.010499545,0.006818632,0.24401398,0.7342914],"study_design_scores_gemma":[0.000080237136,0.000106884894,0.0019705035,0.0002157002,0.000033411852,0.0004345814,0.00017702585,0.0023751566,0.0025048414,0.0021201419,0.9899503,0.000031279716],"about_ca_topic_score_codex":0.003044546,"about_ca_topic_score_gemma":0.001684824,"teacher_disagreement_score":0.085216045,"about_ca_system_score_codex":0.0008764472,"about_ca_system_score_gemma":0.00042588438,"threshold_uncertainty_score":0.12155026},"labels":[],"label_agreement":null},{"id":"W4411721468","doi":"10.54097/7hwf8v07","title":"The Impact of Evolving Regulatory Policies on Content Strategies: A Case Study of Xiaohongshu (Red Note) Bloggers","year":2025,"lang":"en","type":"article","venue":"Academic journal of management and social sciences","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Earl Haig Secondary School","funders":"","keywords":"Content (measure theory); Advertising; Business; Mathematics","score_opus":0.04622755882596701,"score_gpt":0.384249712616817,"score_spread":0.33802215379085,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411721468","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9938253,0.000027793505,0.00026765652,0.00037367755,0.000008234759,0.000048513015,0.000030118337,0.000009754941,0.005408943],"genre_scores_gemma":[0.9925604,0.000096814,0.0011046394,0.00026532335,0.0000099142535,0.00009342852,0.000047511556,0.000018040233,0.005803926],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9983681,0.00088032545,0.000044817603,0.00016339826,0.000252612,0.00029066898],"domain_scores_gemma":[0.9905344,0.006821653,0.0007565023,0.00055098184,0.00056103885,0.0007754596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025831047,0.00022926701,0.00024734347,0.00120099,0.0067443573,0.0025825975,0.00083399133,0.0015329892,0.0030085624],"category_scores_gemma":[0.00621065,0.0002280744,0.0002301075,0.0018689416,0.0029651045,0.0026347172,0.0016983278,0.0013311038,0.00040860716],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000518376,0.0021903408,0.231679,0.00046079446,0.00005945614,0.032610536,0.5892351,0.0019034939,0.012953254,0.017652245,0.008633131,0.10210423],"study_design_scores_gemma":[0.00009167151,0.0007590139,0.1542473,0.000179365,0.00006092643,0.002292048,0.74828255,0.005994566,0.0060902704,0.0036166979,0.078241,0.00014453566],"about_ca_topic_score_codex":0.012956158,"about_ca_topic_score_gemma":0.038479235,"teacher_disagreement_score":0.012956158,"about_ca_system_score_codex":0.003541741,"about_ca_system_score_gemma":0.0028642863,"threshold_uncertainty_score":0.025761545},"labels":[],"label_agreement":null},{"id":"W4412445659","doi":"10.1109/eurocon64445.2025.11073281","title":"Relation on Hesitation in Intuitionistic Fuzzy Sets and Decision Making","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Relation (database); Computer science; Fuzzy set; Artificial intelligence; Fuzzy logic; Mathematics; Data mining","score_opus":0.010991742315215805,"score_gpt":0.32447154964383296,"score_spread":0.31347980732861713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412445659","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3727791,0.0009948646,0.6050119,0.0012752722,0.00015755162,0.00017511904,0.00016067468,0.00020990802,0.019235658],"genre_scores_gemma":[0.97343624,0.00014282107,0.025621371,0.00007110714,0.000045440913,0.00006064277,0.000050199356,0.00000828835,0.0005638683],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9920942,0.0031812836,0.00085018564,0.0011192,0.0023113198,0.00044382835],"domain_scores_gemma":[0.9758664,0.018625505,0.0020484785,0.00084614777,0.002195495,0.00041794116],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0063307243,0.0007392802,0.0006132123,0.0021586162,0.0011279334,0.0027547718,0.00077279255,0.0011092885,0.0021469614],"category_scores_gemma":[0.033089075,0.00027065302,0.0010871569,0.0016665523,0.0032179384,0.0035556906,0.001381341,0.001306302,0.00014390705],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011834992,0.0002433214,0.03259124,0.0009232594,0.0005369373,0.0021487388,0.007659186,0.19697466,0.007813189,0.6325179,0.0020623142,0.11534587],"study_design_scores_gemma":[0.00005554184,0.00036682692,0.010459935,0.00013106136,0.00013118116,0.00057870627,0.0016418303,0.5160266,0.0049946583,0.46287212,0.002566888,0.00017452207],"about_ca_topic_score_codex":0.0012089642,"about_ca_topic_score_gemma":0.0005248887,"teacher_disagreement_score":0.0063307243,"about_ca_system_score_codex":0.0023153818,"about_ca_system_score_gemma":0.0008818179,"threshold_uncertainty_score":0.033480465},"labels":[],"label_agreement":null},{"id":"W4412554184","doi":"10.22214/ijraset.2025.73052","title":"Enhancing Technical Documentation through Intelligent Text Summarization Techniques","year":2025,"lang":"en","type":"article","venue":"International Journal for Research in Applied Science and Engineering Technology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Automatic summarization; Documentation; Computer science; Technical documentation; Information retrieval; Natural language processing; World Wide Web; Programming language","score_opus":0.03231267546222466,"score_gpt":0.425018475019682,"score_spread":0.3927057995574573,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412554184","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.079909645,0.0033885892,0.87157583,0.0009951296,0.00038511067,0.0009995452,0.005564443,0.031762727,0.0054189432],"genre_scores_gemma":[0.16557719,0.0019130514,0.8015342,0.0001822823,0.00030694657,0.0008286092,0.023994504,0.0010252617,0.0046379855],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9973815,0.00095132634,0.00038114813,0.0006236984,0.0005634594,0.00009885047],"domain_scores_gemma":[0.9889671,0.0056091584,0.0014583842,0.0013325987,0.0024719434,0.00016091204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003485153,0.0018299913,0.00092094345,0.0067353654,0.00086908415,0.0025229035,0.0014095684,0.0009354036,0.0026485047],"category_scores_gemma":[0.017809652,0.00044371263,0.0013861563,0.0044650696,0.0005074871,0.003184392,0.0016504531,0.0013415886,0.0034616892],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002483995,0.00016903454,0.0020467981,0.0017086404,0.00011004659,0.00029119107,0.0015440166,0.012031427,0.028391365,0.0020905172,0.01776946,0.93359905],"study_design_scores_gemma":[0.00037591282,0.0016447619,0.015690586,0.0008601087,0.0011631212,0.0014615636,0.0037951188,0.5807056,0.15665878,0.023411779,0.21391694,0.00031573058],"about_ca_topic_score_codex":0.0021530492,"about_ca_topic_score_gemma":0.0035611414,"teacher_disagreement_score":0.0067353654,"about_ca_system_score_codex":0.00065364764,"about_ca_system_score_gemma":0.0014190151,"threshold_uncertainty_score":0.018431485},"labels":[],"label_agreement":null},{"id":"W4412887723","doi":"10.18653/v1/2025.findings-acl.1160","title":"Understanding the Influence of Synthetic Data for Text Embedders","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; Samsung; National Science Foundation","keywords":"Computer science; Information retrieval; Data science","score_opus":0.08669165571493846,"score_gpt":0.34819054553802975,"score_spread":0.2614988898230913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412887723","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.66842806,0.0032781614,0.29307142,0.0035466205,0.0009133064,0.000686263,0.012887807,0.0065931748,0.010595181],"genre_scores_gemma":[0.8269252,0.00072099204,0.13837138,0.00078084326,0.00019212891,0.00070460106,0.028218972,0.00093969156,0.0031461397],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9943777,0.0035590045,0.0003482987,0.00091966026,0.0005936343,0.00020170519],"domain_scores_gemma":[0.95788544,0.029494824,0.0015019391,0.0075864196,0.0028394214,0.0006920001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00902088,0.0014668753,0.00065611105,0.0011890376,0.00068710296,0.0022449552,0.0015145555,0.0017294663,0.002554092],"category_scores_gemma":[0.05826413,0.00057366386,0.00082110614,0.0010833492,0.0016713836,0.004780002,0.0026742725,0.0020797679,0.0020630616],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023140293,0.0012827953,0.048812415,0.0027095005,0.00054179045,0.0008794341,0.0024571277,0.4800454,0.03929755,0.021164551,0.062261913,0.33823353],"study_design_scores_gemma":[0.00019944194,0.00094860356,0.009086793,0.00030964412,0.00009790375,0.00057016994,0.0012167506,0.8876811,0.046404473,0.028725393,0.024653638,0.00010613132],"about_ca_topic_score_codex":0.002149385,"about_ca_topic_score_gemma":0.003453942,"teacher_disagreement_score":0.00902088,"about_ca_system_score_codex":0.00082123716,"about_ca_system_score_gemma":0.00068747863,"threshold_uncertainty_score":0.047707558},"labels":[],"label_agreement":null},{"id":"W4412889548","doi":"10.18653/v1/2025.acl-short.15","title":"Improving the Calibration of Confidence Scores in Text Generation Using the Output Distribution’s Characteristics","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institut de Valorisation des Données; Nvidia","keywords":"Calibration; Computer science; Confidence interval; Statistics; Distribution (mathematics); Mathematics","score_opus":0.024640111490662857,"score_gpt":0.28667247297568743,"score_spread":0.26203236148502457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412889548","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22699797,0.0051274532,0.72533363,0.0019086588,0.0005629196,0.0005809448,0.0057578515,0.026038883,0.0076916637],"genre_scores_gemma":[0.8536234,0.00036232534,0.1328592,0.0005256571,0.00018513728,0.0003674609,0.00941267,0.0014001549,0.0012640307],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98432803,0.0069465805,0.0014504645,0.003807414,0.0027893276,0.0006782302],"domain_scores_gemma":[0.8926799,0.07329357,0.0059107626,0.0146555705,0.011968278,0.0014919101],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02105732,0.002466307,0.0016506936,0.0052768616,0.00097382074,0.004944759,0.0023247832,0.003844985,0.0024210277],"category_scores_gemma":[0.17060837,0.00074167515,0.0012993488,0.0031942893,0.0015067066,0.0062841102,0.003078241,0.0044402634,0.002226517],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022315844,0.0006972505,0.07648181,0.0013754434,0.0008199138,0.00040445116,0.0014237846,0.31425217,0.016911626,0.008981073,0.031238038,0.5451829],"study_design_scores_gemma":[0.00015726314,0.00030602916,0.012250388,0.00015461222,0.0001119995,0.00027435707,0.0002001768,0.94690853,0.016056918,0.018342081,0.005096349,0.00014126585],"about_ca_topic_score_codex":0.003634945,"about_ca_topic_score_gemma":0.0034375077,"teacher_disagreement_score":0.02105732,"about_ca_system_score_codex":0.0018590065,"about_ca_system_score_gemma":0.0015951738,"threshold_uncertainty_score":0.11136311},"labels":[],"label_agreement":null},{"id":"W4413183628","doi":"10.14722/madweb.2025.23004","title":"Evaluating the Strength and Availability of Multilingual Passphrase Authentication","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Authentication (law); Computer security","score_opus":0.033463372864654374,"score_gpt":0.3962153046769616,"score_spread":0.3627519318123072,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413183628","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9957463,0.00019004392,0.001052347,0.000046927955,0.000009883699,0.000034787234,0.00028575942,0.0001380175,0.0024959932],"genre_scores_gemma":[0.9984761,0.000060673006,0.00077738875,0.000014682293,0.0000059212457,0.000010467745,0.00026436863,0.000010529339,0.00037992463],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.9953565,0.0010922439,0.0008138203,0.0004602915,0.0019459975,0.0003311513],"domain_scores_gemma":[0.93596387,0.032688256,0.016151056,0.0044876244,0.0073417365,0.0033675067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034161003,0.00053007685,0.00045165105,0.0024849747,0.00041156786,0.001677387,0.0005247259,0.00074953685,0.0028007857],"category_scores_gemma":[0.042234518,0.00021039,0.00047315564,0.0010823455,0.0006268319,0.0031393345,0.0014251847,0.0007092421,0.0014287859],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002373558,0.00054639095,0.9032566,0.0006003123,0.0003280593,0.00041691953,0.0019547811,0.0025047124,0.009409787,0.0005441196,0.0010274351,0.0770374],"study_design_scores_gemma":[0.00004432987,0.002876098,0.9527387,0.00015287455,0.00020412149,0.0016835247,0.0028889813,0.022616914,0.012967826,0.0006139641,0.0030704385,0.000142246],"about_ca_topic_score_codex":0.0017691862,"about_ca_topic_score_gemma":0.0016228596,"teacher_disagreement_score":0.0034161003,"about_ca_system_score_codex":0.00037803466,"about_ca_system_score_gemma":0.00035373154,"threshold_uncertainty_score":0.018066287},"labels":[],"label_agreement":null},{"id":"W4413367878","doi":"10.18280/mmep.120703","title":"Twitter Sentiment Analysis via Chaotic Quantum Fruit Fly Optimization: Enhancing Feature Selection and Classification","year":2025,"lang":"en","type":"article","venue":"Mathematical Modelling and Engineering Problems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Chaotic; Feature selection; Selection (genetic algorithm); Computer science; Quantum; Sentiment analysis; Feature (linguistics); Artificial intelligence; On the fly; Pattern recognition (psychology); Machine learning; Physics","score_opus":0.015239658600163863,"score_gpt":0.23862027633162247,"score_spread":0.22338061773145862,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413367878","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1415769,0.0003811402,0.85360146,0.00035806742,0.000070964656,0.00013462295,0.00018290343,0.0011179048,0.002576135],"genre_scores_gemma":[0.77068937,0.00014944091,0.22648583,0.00018203659,0.000059524835,0.0001384007,0.00043948076,0.00009834673,0.0017574027],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997216,0.000066099856,0.000019824152,0.000056338948,0.00009957859,0.000036506175],"domain_scores_gemma":[0.9996649,0.00011564904,0.00005200312,0.000027221313,0.00012439568,0.000015837566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006975942,0.0006404912,0.0006890948,0.0007341663,0.00027585344,0.00064196525,0.0005532736,0.00046428738,0.00086919183],"category_scores_gemma":[0.0015699129,0.00017549239,0.00060260936,0.0004885484,0.00024127949,0.00065520254,0.00044092198,0.0003613609,0.00032474342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038257413,0.00019136617,0.010849849,0.00016354761,0.00019160594,0.00018828183,0.00015262615,0.2858702,0.05876294,0.0052469173,0.005374636,0.63262546],"study_design_scores_gemma":[0.000010110159,0.000041907162,0.0011286963,0.0000036041113,0.000010574943,0.000025541516,0.00001671484,0.99307996,0.00404325,0.0009175692,0.00071392005,0.000008062697],"about_ca_topic_score_codex":0.002034086,"about_ca_topic_score_gemma":0.0020991217,"teacher_disagreement_score":0.002034086,"about_ca_system_score_codex":0.0003555296,"about_ca_system_score_gemma":0.00039000218,"threshold_uncertainty_score":0.004044473},"labels":[],"label_agreement":null},{"id":"W4413677204","doi":"10.1016/j.ipm.2025.104335","title":"Class-Missing Semi-supervised document key information extraction via synergistic refinement estimation","year":2025,"lang":"en","type":"article","venue":"Information Processing & Management","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Vector Institute; Western University","funders":"Key Research and Development Projects of Shaanxi Province; National Natural Science Foundation of China","keywords":"Key (lock); Estimation; Class (philosophy); Computer science; Extraction (chemistry); Information retrieval; Artificial intelligence; Engineering; Chemistry; Chromatography; Computer security","score_opus":0.005924988358388565,"score_gpt":0.27765318649217363,"score_spread":0.27172819813378507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413677204","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013002676,0.0010366173,0.9759982,0.00021488684,0.00009720057,0.00015149645,0.00082280405,0.0076366,0.0010395307],"genre_scores_gemma":[0.16212098,0.0006284386,0.8242917,0.00028502734,0.00017668448,0.00029903543,0.006984365,0.00065615424,0.0045576524],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99624383,0.00070408883,0.00032588962,0.0013292049,0.0011245481,0.0002724349],"domain_scores_gemma":[0.9918424,0.0025160175,0.0008418583,0.002634356,0.0019737848,0.0001916554],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035826967,0.0016541894,0.0022610698,0.003037438,0.001026638,0.0019564051,0.0031743753,0.0016580928,0.002128019],"category_scores_gemma":[0.0105234645,0.0007093891,0.0016401933,0.0031804708,0.001294407,0.0045357593,0.003160132,0.0025454417,0.0036915427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053674745,0.00016395558,0.0020611375,0.00047174195,0.00013101369,0.00018611923,0.00033050645,0.022846144,0.03352722,0.0042833723,0.017017405,0.9184446],"study_design_scores_gemma":[0.000088802226,0.00018173482,0.002032684,0.00006527229,0.0001324226,0.0005868532,0.00027661983,0.90650314,0.05181937,0.020142484,0.01808201,0.00008859321],"about_ca_topic_score_codex":0.002715856,"about_ca_topic_score_gemma":0.0059215548,"teacher_disagreement_score":0.0035826967,"about_ca_system_score_codex":0.00078752823,"about_ca_system_score_gemma":0.0020831442,"threshold_uncertainty_score":0.018947363},"labels":[],"label_agreement":null},{"id":"W4413975987","doi":"10.4018/ijisss.388002","title":"LLM-Guided Multimodal Information Fusion With Hierarchical Spatio-Temporal Graph Network for Sentiment Analysis","year":2025,"lang":"en","type":"article","venue":"International Journal of Information Systems in the Service Sector","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Graph; Information fusion; Sentiment analysis; Artificial intelligence; Fusion; Data mining; Theoretical computer science","score_opus":0.009639930156432422,"score_gpt":0.2799848287921413,"score_spread":0.2703448986357089,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413975987","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017502056,0.00026068708,0.9795777,0.00026343923,0.000053974087,0.00003590464,0.00012207955,0.0006692129,0.0015149127],"genre_scores_gemma":[0.75533515,0.00043983554,0.23589809,0.00041739206,0.00011393184,0.00016141713,0.0007874724,0.00026394802,0.0065827416],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997004,0.00008596186,0.000014103647,0.00009292601,0.00006365872,0.00004302085],"domain_scores_gemma":[0.999686,0.00013525866,0.000046857724,0.00003736023,0.0000736095,0.000020967424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00062980177,0.0009732294,0.00057239,0.0008504928,0.00036581664,0.0006057321,0.0010366406,0.0009398584,0.0020685242],"category_scores_gemma":[0.0015770249,0.0003242491,0.0010868328,0.0008068498,0.00054932287,0.0015580322,0.0011598493,0.0012082579,0.00059891335],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026314004,0.00011482709,0.0013299492,0.00012752975,0.00014990036,0.00020667182,0.00022975352,0.68942046,0.02845704,0.022686612,0.0053681615,0.25164598],"study_design_scores_gemma":[0.0000015425718,0.000009836479,0.000075556294,0.0000019656231,0.000006381465,0.0000073039764,0.000006024761,0.99574584,0.00089316565,0.0029842528,0.0002648622,0.000003299704],"about_ca_topic_score_codex":0.006205805,"about_ca_topic_score_gemma":0.006858183,"teacher_disagreement_score":0.006205805,"about_ca_system_score_codex":0.0010421675,"about_ca_system_score_gemma":0.00052524277,"threshold_uncertainty_score":0.012339354},"labels":[],"label_agreement":null},{"id":"W4414359746","doi":"10.24963/ijcai.2025/816","title":"Logic Distillation: Learning from Code Function by Function for Decision-making Tasks","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Function (biology); Variety (cybernetics); Code (set theory); Comprehension; Knowledge base; Base (topology)","score_opus":0.014202416373039149,"score_gpt":0.30295157743381,"score_spread":0.28874916106077086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414359746","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020641062,0.00048489365,0.96657056,0.0006485717,0.000081194914,0.00018347321,0.00043569255,0.0078516975,0.0031028325],"genre_scores_gemma":[0.4592866,0.00045129823,0.5317728,0.00080796453,0.00010118371,0.00048337714,0.002043709,0.0005457241,0.0045074206],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992156,0.00022682677,0.00005113212,0.00024627306,0.00016925794,0.00009093449],"domain_scores_gemma":[0.9984188,0.0010033726,0.000101500154,0.0002363197,0.00014754754,0.00009238522],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011135702,0.0017307213,0.00091854844,0.00092840526,0.0004759434,0.0014435903,0.002493872,0.0013167077,0.0071440237],"category_scores_gemma":[0.006512446,0.00061060075,0.0012719829,0.00063151575,0.0009559318,0.002773548,0.0022831918,0.002894699,0.0020798075],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040815357,0.00035019367,0.002355489,0.00047075955,0.00010568313,0.00024607967,0.00026094713,0.2476398,0.009781179,0.027152667,0.009632173,0.7015968],"study_design_scores_gemma":[0.000028972569,0.000059259357,0.0001319721,0.000030049088,0.000017158658,0.000033895856,0.000024562522,0.964981,0.003564945,0.028821724,0.002293538,0.000012825431],"about_ca_topic_score_codex":0.003950248,"about_ca_topic_score_gemma":0.006694846,"teacher_disagreement_score":0.0071440237,"about_ca_system_score_codex":0.0010462123,"about_ca_system_score_gemma":0.0019854438,"threshold_uncertainty_score":0.023899198},"labels":[],"label_agreement":null},{"id":"W4414588928","doi":"10.3390/app151910506","title":"Mind the Link: Discourse Link-Aware Hallucination Detection in Summarization","year":2025,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Privy Council Office","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Iran Telecommunication Research Center; National Research Foundation of Korea; National Research Foundation","keywords":"Automatic summarization; Meaning (existential); Representation (politics); Relation (database); Content (measure theory); Semantic relation; Structuring","score_opus":0.01175317348728903,"score_gpt":0.30321453504426277,"score_spread":0.2914613615569737,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414588928","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.113292515,0.0026064182,0.8296811,0.00069679064,0.00024606017,0.0008681348,0.0032790944,0.04466518,0.004664613],"genre_scores_gemma":[0.43378684,0.0005724637,0.55656564,0.00021887738,0.00018211424,0.00033866233,0.0041732546,0.0005545324,0.0036076664],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988992,0.00032549212,0.000111994785,0.0003091597,0.00028278193,0.00007118421],"domain_scores_gemma":[0.99489,0.002206768,0.00091635017,0.00055901957,0.0012189371,0.00020897872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016989745,0.001182745,0.00065888517,0.004399557,0.00056961516,0.0017311967,0.0010695348,0.0011085712,0.0032267636],"category_scores_gemma":[0.012954402,0.0002732419,0.00047146573,0.0015841933,0.00040943222,0.0033228134,0.0017108422,0.0008429933,0.0016341927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001008675,0.00018749476,0.0068848697,0.0010522063,0.00019352048,0.0004102981,0.0015818182,0.0050141257,0.045179658,0.0031408598,0.0142624695,0.92108405],"study_design_scores_gemma":[0.0002555282,0.0017662497,0.031830177,0.0003591021,0.0007274838,0.0016517799,0.0031771066,0.68765384,0.18430875,0.03137453,0.05659986,0.00029562743],"about_ca_topic_score_codex":0.0014943093,"about_ca_topic_score_gemma":0.0020066497,"teacher_disagreement_score":0.004399557,"about_ca_system_score_codex":0.00044096302,"about_ca_system_score_gemma":0.0005849707,"threshold_uncertainty_score":0.01079458},"labels":[],"label_agreement":null},{"id":"W4415203880","doi":"10.5753/jbcs.2025.5815","title":"Cross-Lingual Keyword Extraction for Pesticide Terminology in Brazilian Portuguese and English","year":2025,"lang":"en","type":"article","venue":"Journal of the Brazilian Computer Society","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec en Outaouais; Institut National de la Recherche Scientifique; Université du Québec à Montréal","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico; Fundação de Amparo à Pesquisa do Estado de São Paulo; Universität St. Gallen","keywords":"Terminology; Portuguese; Standardization; Brazilian Portuguese; Pesticide; Representation (politics)","score_opus":0.009630347366461907,"score_gpt":0.3190914924121837,"score_spread":0.3094611450457218,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415203880","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8063482,0.011732407,0.07871577,0.002133887,0.00044260878,0.0011218381,0.06531005,0.0050904476,0.029104754],"genre_scores_gemma":[0.7689124,0.0032157593,0.12814459,0.00028792483,0.00013449976,0.000744558,0.093833625,0.0008326127,0.003894098],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983551,0.00054593047,0.00038662975,0.0003711865,0.00024264617,0.00009838033],"domain_scores_gemma":[0.99371827,0.0041555697,0.00042416572,0.00043220219,0.0011463085,0.00012334489],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001540768,0.00093702576,0.00061970134,0.0046676407,0.0009791016,0.0011039833,0.0004600686,0.00046866926,0.003250073],"category_scores_gemma":[0.010354368,0.00037178554,0.00082696613,0.0036516055,0.0005154219,0.0016774886,0.0013948021,0.0005473141,0.0014960999],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014360396,0.00048028945,0.024598107,0.019423816,0.00032325718,0.010352968,0.02269615,0.009944714,0.18314046,0.009902694,0.049872607,0.66782886],"study_design_scores_gemma":[0.00045186703,0.00064570055,0.16856676,0.0024937186,0.0009939886,0.014824426,0.027736628,0.10078257,0.15442517,0.0065463344,0.52207744,0.00045536546],"about_ca_topic_score_codex":0.02028047,"about_ca_topic_score_gemma":0.029931124,"teacher_disagreement_score":0.02028047,"about_ca_system_score_codex":0.0011294085,"about_ca_system_score_gemma":0.002297055,"threshold_uncertainty_score":0.040324867},"labels":[],"label_agreement":null},{"id":"W4415273246","doi":"10.1007/978-981-95-3462-3_5","title":"Analysis of Metadata and Cough Signal Complexities in Tuberculosis Screening","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Mila - Quebec Artificial Intelligence Institute; Centre Hospitalier de l’Université de Montréal; Université TÉLUQ","funders":"","keywords":"Metadata; Preprocessor; Hurst exponent; Linear discriminant analysis; Time series; Data pre-processing; SIGNAL (programming language); Novelty","score_opus":0.0229232635157959,"score_gpt":0.2808472809139578,"score_spread":0.2579240173981619,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415273246","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8569029,0.0050011496,0.12228377,0.0012330788,0.00018478885,0.0001569101,0.0061856275,0.001350977,0.0067007877],"genre_scores_gemma":[0.96022344,0.00084400707,0.03295989,0.00006007204,0.00017820126,0.000049589722,0.0035922644,0.000099058496,0.0019934243],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99871457,0.00030783052,0.00016417658,0.00022094902,0.00044875272,0.00014380924],"domain_scores_gemma":[0.97933793,0.016438551,0.0016205973,0.00072467886,0.0014437768,0.0004344654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017786067,0.00045049054,0.0005356779,0.0036968798,0.000556284,0.0022993346,0.0005699462,0.00076935306,0.0024669562],"category_scores_gemma":[0.01688742,0.0002868976,0.00062041817,0.0028983308,0.00050237286,0.0025737786,0.0010345668,0.0008151578,0.0009987359],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031333652,0.0004409435,0.35532355,0.00084034924,0.000278844,0.0018044843,0.0010690598,0.044390015,0.048028655,0.011964717,0.0071426206,0.5255834],"study_design_scores_gemma":[0.000044133336,0.000619531,0.28294307,0.00023813336,0.00046499254,0.0034438963,0.0022380664,0.6409763,0.032561734,0.027834361,0.008472852,0.00016297604],"about_ca_topic_score_codex":0.003596238,"about_ca_topic_score_gemma":0.004356856,"teacher_disagreement_score":0.0036968798,"about_ca_system_score_codex":0.0008826753,"about_ca_system_score_gemma":0.00090437225,"threshold_uncertainty_score":0.009406269},"labels":[],"label_agreement":null},{"id":"W4415373758","doi":"10.1108/ejm-04-2025-0301","title":"Understanding the defining characteristics of a high-quality conceptual article","year":2025,"lang":"en","type":"article","venue":"European Journal of Marketing","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University","funders":"","keywords":"Operationalization; Rubric; Conceptual framework; Conceptual model; Quality (philosophy); Empirical research; The Conceptual Framework","score_opus":0.052879564351633386,"score_gpt":0.2909858425233979,"score_spread":0.2381062781717645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415373758","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6683178,0.019387035,0.16136773,0.04244797,0.0018628011,0.0032205542,0.0019128686,0.000590861,0.10089244],"genre_scores_gemma":[0.9352081,0.0032715371,0.05537823,0.0011665627,0.00054599514,0.0010605782,0.00075208046,0.00017964751,0.0024374032],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9330503,0.028886525,0.012828682,0.0031212012,0.020221807,0.0018913888],"domain_scores_gemma":[0.35702026,0.41606602,0.06962955,0.021332033,0.123485215,0.0124669],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07611896,0.00055171375,0.001025269,0.015403031,0.0032802538,0.025654081,0.001840157,0.0026287332,0.0042879106],"category_scores_gemma":[0.40228134,0.0005837915,0.0010332813,0.015042127,0.009047835,0.019658936,0.0066312333,0.0022798846,0.00091554574],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039548604,0.00036008898,0.17192166,0.015094606,0.0005544341,0.0025938672,0.19614609,0.0017895694,0.0076400135,0.1820487,0.019949976,0.40150556],"study_design_scores_gemma":[0.00016835138,0.00078452076,0.1887824,0.017305385,0.0007849163,0.006279974,0.27167368,0.0068479995,0.00856367,0.24665397,0.2515195,0.00063570455],"about_ca_topic_score_codex":0.0013277631,"about_ca_topic_score_gemma":0.001642607,"teacher_disagreement_score":0.92388105,"about_ca_system_score_codex":0.006859804,"about_ca_system_score_gemma":0.011396958,"threshold_uncertainty_score":0.40256035},"labels":[],"label_agreement":null},{"id":"W4415422573","doi":"10.1016/j.eswa.2025.130088","title":"DGSEP: Dual-stage generative model with sequence-oriented labeling and element-to-tuple prompting improves aspect sentiment triplet extraction","year":2025,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Sichuan Province Science and Technology Support Program; Key Research and Development Program of Sichuan Province; Chengdu Science and Technology Bureau; Department of Science and Technology of Sichuan Province; National Natural Science Foundation of China","keywords":"Fuse (electrical); Generative grammar; Generative model; Sequence (biology); Task (project management); Sentiment analysis","score_opus":0.018136075905115134,"score_gpt":0.32024835108470256,"score_spread":0.30211227517958744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415422573","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01786525,0.00028933483,0.95478874,0.00034246512,0.0002394179,0.00020948304,0.0016924912,0.021426477,0.0031462114],"genre_scores_gemma":[0.30761775,0.00026304633,0.6644517,0.0006992757,0.00019038646,0.00036336918,0.010623958,0.0040194574,0.011770983],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99923146,0.00018585497,0.000042484135,0.0003083419,0.00014703446,0.00008484006],"domain_scores_gemma":[0.9986131,0.00065214565,0.000058139798,0.00032378055,0.00027924648,0.000073581505],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010692817,0.0015775168,0.00095123105,0.0012231977,0.0007370836,0.001517908,0.00201975,0.0017297461,0.008648429],"category_scores_gemma":[0.0028896497,0.00096006325,0.0019464382,0.0011280844,0.0005535012,0.0023430798,0.002244402,0.0029227575,0.0060673202],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011349309,0.0005043329,0.0057381596,0.0005623579,0.00025485523,0.00082826184,0.00084848714,0.09650544,0.05852775,0.021240566,0.05593546,0.75791955],"study_design_scores_gemma":[0.00004192774,0.000057110592,0.000635336,0.000023515875,0.00006366312,0.000117860494,0.0000638951,0.96502584,0.010751009,0.015424658,0.0077646747,0.000030430989],"about_ca_topic_score_codex":0.007068081,"about_ca_topic_score_gemma":0.016884714,"teacher_disagreement_score":0.008648429,"about_ca_system_score_codex":0.00065910345,"about_ca_system_score_gemma":0.001874743,"threshold_uncertainty_score":0.028931856},"labels":[],"label_agreement":null},{"id":"W4415598779","doi":"10.1115/detc2025-163043","title":"Cognitive Demands and Individual Differences in Product Similarity Judgements: A Context-Dependent Model","year":2025,"lang":"","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Similarity (geometry); Cognition; Context (archaeology); Product (mathematics); Schema (genetic algorithms); Pairwise comparison","score_opus":0.04925063824614321,"score_gpt":0.3173307582133334,"score_spread":0.26808011996719017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415598779","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99179476,0.000065235174,0.0054654307,0.00007093256,0.000011597857,0.00009069558,0.00010959388,0.000020095851,0.0023716185],"genre_scores_gemma":[0.9966689,0.000024961597,0.0027122328,0.000027234726,0.000008752656,0.000077280274,0.00008375026,0.000016601009,0.00038025973],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972658,0.0012424098,0.00012769851,0.00061286223,0.0005765699,0.00017469772],"domain_scores_gemma":[0.93532777,0.049886424,0.0052060313,0.0055472166,0.002541696,0.0014907693],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00528054,0.00055499247,0.00064022787,0.000998988,0.00041566096,0.0021809714,0.001020476,0.0012044504,0.005944094],"category_scores_gemma":[0.04943313,0.0005722855,0.00093669724,0.000820602,0.0012869653,0.0014372666,0.0015519084,0.0011197936,0.00055978564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.015736245,0.008705946,0.6662231,0.0013984957,0.0017176808,0.0012964128,0.0144915525,0.07079271,0.11598409,0.022630118,0.0018767728,0.07914684],"study_design_scores_gemma":[0.00034573223,0.0016891644,0.78795314,0.000054155156,0.00043188885,0.00035365368,0.0010710551,0.18766056,0.0048966985,0.014442845,0.0009156481,0.00018547445],"about_ca_topic_score_codex":0.0024803882,"about_ca_topic_score_gemma":0.0016652237,"teacher_disagreement_score":0.005944094,"about_ca_system_score_codex":0.0006441756,"about_ca_system_score_gemma":0.00040320968,"threshold_uncertainty_score":0.027926505},"labels":[],"label_agreement":null},{"id":"W65121486","doi":"10.1007/978-3-642-23863-5_1","title":"Autonomous and Adaptive Identification of Topics in Unstructured Text","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Computer science; Novelty; Identification (biology); Flexibility (engineering); Artificial intelligence; Process (computing); Natural language processing; Information retrieval; Programming language","score_opus":0.016527156940934352,"score_gpt":0.2503990093970874,"score_spread":0.23387185245615308,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W65121486","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1636414,0.0011567755,0.8231285,0.00033789902,0.00019683084,0.0002061847,0.000696039,0.0048195194,0.0058168992],"genre_scores_gemma":[0.6155721,0.0006758151,0.3696276,0.00014350007,0.0002878411,0.00016578399,0.0017177635,0.0005171743,0.011292399],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99944896,0.00011322117,0.000029019508,0.00022856281,0.00012593964,0.0000542117],"domain_scores_gemma":[0.99825364,0.0010916683,0.00012496543,0.00017959515,0.0002492531,0.000100872865],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00084683317,0.00047552193,0.00056976994,0.0012958674,0.0004105226,0.0012656153,0.000868702,0.0007253765,0.0019348918],"category_scores_gemma":[0.003271205,0.00028331557,0.00037325386,0.0009960984,0.00050805334,0.0019094187,0.0015229979,0.0006538274,0.0019321892],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006215168,0.00017619804,0.0061475914,0.00028656088,0.000060876155,0.00017723044,0.0009313399,0.01101668,0.17601635,0.003190545,0.0067791073,0.7945961],"study_design_scores_gemma":[0.00004502261,0.00023484796,0.014553595,0.000041538486,0.00009612723,0.00038382638,0.00088671193,0.88218725,0.0749473,0.014512675,0.012069314,0.000041892636],"about_ca_topic_score_codex":0.0010853921,"about_ca_topic_score_gemma":0.0019603886,"teacher_disagreement_score":0.0019348918,"about_ca_system_score_codex":0.0002702675,"about_ca_system_score_gemma":0.0004226841,"threshold_uncertainty_score":0.006472826},"labels":[],"label_agreement":null},{"id":"W67066386","doi":"","title":"A hybrid approach to the identification and expansion of abbreviations","year":2000,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Task (project management); Class (philosophy); Natural language processing; Identification (biology); Word (group theory); Artificial intelligence; Domain (mathematical analysis); Information retrieval; Engineering","score_opus":0.013420753987818133,"score_gpt":0.2619721861350616,"score_spread":0.24855143214724348,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W67066386","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0066759842,0.00044414867,0.9777379,0.00027947666,0.00014609766,0.00023362198,0.0003279182,0.010731986,0.003422886],"genre_scores_gemma":[0.03376412,0.00020787024,0.9597787,0.00023512392,0.000088992,0.00020993901,0.0006361997,0.0004053538,0.004673717],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9960776,0.00077051495,0.00037543094,0.0014296598,0.0011998458,0.000146824],"domain_scores_gemma":[0.9933565,0.002662174,0.00033238082,0.0014881734,0.0019556487,0.00020509362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024132323,0.0010922332,0.0014921735,0.0042716037,0.0009656565,0.003291457,0.003112504,0.0019255403,0.008479757],"category_scores_gemma":[0.0093174,0.0006689213,0.0016062259,0.0032668095,0.0011650765,0.0050207875,0.0029880463,0.0017537293,0.0065828054],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019197146,0.00013736483,0.0015637272,0.00035380517,0.00011258352,0.00021746123,0.00073577126,0.005125419,0.044780295,0.012009053,0.008185469,0.926587],"study_design_scores_gemma":[0.0002002381,0.0007619656,0.004123117,0.00024939433,0.0005050886,0.0021477179,0.0009001143,0.5918881,0.097402245,0.058211323,0.24316812,0.00044254903],"about_ca_topic_score_codex":0.0031270625,"about_ca_topic_score_gemma":0.004300721,"teacher_disagreement_score":0.008479757,"about_ca_system_score_codex":0.00074011827,"about_ca_system_score_gemma":0.0014768464,"threshold_uncertainty_score":0.028367579},"labels":[],"label_agreement":null},{"id":"W6911926465","doi":"10.5281/zenodo.14137274","title":"A survey on audio analysis: Text characterization and summarization","year":2024,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Automatic summarization; Search engine indexing; Context (archaeology); Multi-document summarization; Transcription (linguistics); Variety (cybernetics)","score_opus":0.03022118782953534,"score_gpt":0.2682771205464275,"score_spread":0.23805593271689215,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6911926465","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019755825,0.32190257,0.5997932,0.0034971344,0.001683084,0.00073723734,0.0073557394,0.01088867,0.03438659],"genre_scores_gemma":[0.072783425,0.41449758,0.44993046,0.0021050158,0.0036038533,0.0012541522,0.024776127,0.0032719725,0.027777322],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970902,0.0007413055,0.00033592441,0.0006348268,0.0010775188,0.00012024013],"domain_scores_gemma":[0.9889043,0.0077145235,0.00055446394,0.0007384865,0.0019427816,0.00014547458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021116151,0.0016643243,0.0014747116,0.009377017,0.0005481066,0.0026562384,0.0017649679,0.0011442363,0.008332787],"category_scores_gemma":[0.009945608,0.0005215433,0.0011889532,0.010189342,0.0006564369,0.003618352,0.0008694805,0.00101467,0.00862135],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006924249,0.000047401103,0.00071593124,0.004177914,0.00005655861,0.00006902137,0.00027151886,0.0008502486,0.00890536,0.001737178,0.021858022,0.96124166],"study_design_scores_gemma":[0.000058516303,0.0005012949,0.01353104,0.003920752,0.0003662999,0.0024148135,0.002724398,0.026347086,0.063884206,0.016936507,0.8690772,0.0002378739],"about_ca_topic_score_codex":0.0012743027,"about_ca_topic_score_gemma":0.0011744844,"teacher_disagreement_score":0.009377017,"about_ca_system_score_codex":0.0005143642,"about_ca_system_score_gemma":0.0009998879,"threshold_uncertainty_score":0.02787596},"labels":[],"label_agreement":null},{"id":"W6930397270","doi":"10.5281/zenodo.12485825","title":"hanstone quartz care and maintenance pdf","year":2024,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Work (physics); Point (geometry); Product (mathematics); Troubleshooting; Context (archaeology)","score_opus":0.014391820200062227,"score_gpt":0.24995228027409008,"score_spread":0.23556046007402787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6930397270","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001714109,0.0013358918,0.009077756,0.0019199033,0.0033087896,0.00052572193,0.011151372,0.017237982,0.95372844],"genre_scores_gemma":[0.0020200252,0.00041669916,0.0017708067,0.00028123223,0.00020853058,0.00007320747,0.002495802,0.002562315,0.9901714],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988422,0.000040287836,0.000042528216,0.00015830909,0.0008441947,0.00007247594],"domain_scores_gemma":[0.99701416,0.0003016317,0.00009142555,0.00051699474,0.0017467795,0.00032901354],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0009350841,0.0011210355,0.0009774505,0.0031066283,0.0020932427,0.003747259,0.0021657755,0.001883158,0.83500093],"category_scores_gemma":[0.004513185,0.00094274036,0.00065221643,0.0022051055,0.0006845546,0.0033270433,0.0025316037,0.0019114118,0.7313821],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000050567745,0.000033737113,0.000107914086,0.00019259959,0.000003250324,0.00009144986,0.00007425921,0.00006295875,0.002889771,0.0019070423,0.9069409,0.08764566],"study_design_scores_gemma":[0.000011383826,0.000015254699,0.00032490442,0.00002986409,0.0000027366425,0.00011822453,0.000043711472,0.000050939667,0.0013170491,0.00037212035,0.9977053,0.00000852832],"about_ca_topic_score_codex":0.003505262,"about_ca_topic_score_gemma":0.0076392703,"teacher_disagreement_score":0.16499907,"about_ca_system_score_codex":0.0011209955,"about_ca_system_score_gemma":0.0016722621,"threshold_uncertainty_score":0.23535115},"labels":[],"label_agreement":null},{"id":"W6958505310","doi":"10.6084/m9.figshare.6999395","title":"Additional file 3: of The prevalence of low back pain in the emergency department: a descriptive study set in the Charles V. Keating Emergency and Trauma Centre, Halifax, Nova Scotia, Canada","year":2018,"lang":"en","type":"article","venue":"Figshare","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Low back pain; Back pain; Minimum Data Set; Emergency department; Low income","score_opus":0.031378791300347444,"score_gpt":0.2582453842967842,"score_spread":0.22686659299643674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6958505310","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011088725,0.000011639556,0.000048725407,0.00004049491,0.000006229065,0.0000935332,0.99798787,0.000022021995,0.0006805046],"genre_scores_gemma":[0.048739452,0.00020745084,0.0013234579,0.00030265868,0.00004048532,0.0035877484,0.9376172,0.00017686294,0.008004647],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9992329,0.000084718085,0.0001669762,0.00013464535,0.00020928192,0.00017140704],"domain_scores_gemma":[0.98779863,0.005773317,0.00137116,0.00070184015,0.0038186635,0.00053639075],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0009845463,0.00071818003,0.0011691889,0.0030775869,0.0013122564,0.0011536541,0.001542274,0.0006280072,0.4792305],"category_scores_gemma":[0.017515784,0.00048286052,0.00092209893,0.0061727674,0.00035543236,0.0009986968,0.0007286185,0.00072820723,0.028298475],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036114984,0.00011999569,0.043860372,0.0023613437,0.00009945734,0.00015420129,0.0003938231,0.0005808916,0.000078991725,0.00052599463,0.9449143,0.006549506],"study_design_scores_gemma":[0.0030339079,0.00021926014,0.7226945,0.0065587685,0.00030739667,0.000760141,0.0034620522,0.0024259922,0.000546723,0.0018257353,0.25792977,0.00023581758],"about_ca_topic_score_codex":0.52157027,"about_ca_topic_score_gemma":0.49261644,"teacher_disagreement_score":0.4792305,"about_ca_system_score_codex":0.0044151763,"about_ca_system_score_gemma":0.008342983,"threshold_uncertainty_score":0.9624946},"labels":[],"label_agreement":null},{"id":"W6968187110","doi":"10.5281/zenodo.15745479","title":"Database of economic valuation of river ecosystem services stemming from ecohidrological forest management in Catalonia (Spain)","year":2025,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Valuation (finance); Ecosystem services; Forest management; Ecosystem; Forest ecology; Ecosystem management; Weighting","score_opus":0.028155607762459022,"score_gpt":0.2720776059287713,"score_spread":0.24392199816631227,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6968187110","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018131712,0.0001260619,0.00009404505,0.000039183065,0.000018591527,0.000025007997,0.99709606,0.00014212844,0.00064583134],"genre_scores_gemma":[0.0021374277,0.0000744233,0.00034730538,0.000018486773,0.0000055360183,0.000117406875,0.9968899,0.000018857254,0.00039084244],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99905163,0.00018045786,0.00016960631,0.0002530146,0.00021373325,0.00013147133],"domain_scores_gemma":[0.9978033,0.0005510747,0.00030478515,0.0003977937,0.0007323174,0.00021079942],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011195558,0.0014695906,0.0010370753,0.0040305555,0.00035394082,0.0018018563,0.0018579146,0.0014356942,0.017760215],"category_scores_gemma":[0.0047195605,0.00034976288,0.0008008559,0.0052617034,0.00027026134,0.0005258183,0.0011425965,0.0009218763,0.017840264],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029404415,0.00011332108,0.010190683,0.0021763325,0.00012185006,0.00016754497,0.00011623201,0.0017693728,0.00035991464,0.0009486132,0.971218,0.012524006],"study_design_scores_gemma":[0.0005134271,0.00007055381,0.08334345,0.0010507432,0.000106985135,0.00022139988,0.0005279658,0.0030021025,0.0007474363,0.0013306786,0.90900487,0.00008037376],"about_ca_topic_score_codex":0.044132255,"about_ca_topic_score_gemma":0.040325623,"teacher_disagreement_score":0.044132255,"about_ca_system_score_codex":0.0019661319,"about_ca_system_score_gemma":0.002029814,"threshold_uncertainty_score":0.08775079},"labels":[],"label_agreement":null},{"id":"W6968317102","doi":"10.5281/zenodo.14137275","title":"A survey on audio analysis: Text characterization and summarization","year":2024,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Automatic summarization; Search engine indexing; Context (archaeology); Multi-document summarization; Transcription (linguistics); Variety (cybernetics)","score_opus":0.03022118782953534,"score_gpt":0.2682771205464275,"score_spread":0.23805593271689215,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6968317102","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019755825,0.32190257,0.5997932,0.0034971344,0.001683084,0.00073723734,0.0073557394,0.01088867,0.03438659],"genre_scores_gemma":[0.072783425,0.41449758,0.44993046,0.0021050158,0.0036038533,0.0012541522,0.024776127,0.0032719725,0.027777322],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970902,0.0007413055,0.00033592441,0.0006348268,0.0010775188,0.00012024013],"domain_scores_gemma":[0.9889043,0.0077145235,0.00055446394,0.0007384865,0.0019427816,0.00014547458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021116151,0.0016643243,0.0014747116,0.009377017,0.0005481066,0.0026562384,0.0017649679,0.0011442363,0.008332787],"category_scores_gemma":[0.009945608,0.0005215433,0.0011889532,0.010189342,0.0006564369,0.003618352,0.0008694805,0.00101467,0.00862135],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006924249,0.000047401103,0.00071593124,0.004177914,0.00005655861,0.00006902137,0.00027151886,0.0008502486,0.00890536,0.001737178,0.021858022,0.96124166],"study_design_scores_gemma":[0.000058516303,0.0005012949,0.01353104,0.003920752,0.0003662999,0.0024148135,0.002724398,0.026347086,0.063884206,0.016936507,0.8690772,0.0002378739],"about_ca_topic_score_codex":0.0012743027,"about_ca_topic_score_gemma":0.0011744844,"teacher_disagreement_score":0.009377017,"about_ca_system_score_codex":0.0005143642,"about_ca_system_score_gemma":0.0009998879,"threshold_uncertainty_score":0.02787596},"labels":[],"label_agreement":null},{"id":"W6968664339","doi":"10.5281/zenodo.3965653","title":"A SEMANTIC METADATA ENRICHMENT SOFTWARE ECOSYSTEM BASED ON TOPIC METADATA ENRICHMENTS","year":2020,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Metadata; Semantic grid; Metadata repository; Metadata modeling; Meta Data Services; Semantic computing; Geospatial metadata; Data element; Database catalog","score_opus":0.04275775834215196,"score_gpt":0.2600206395142916,"score_spread":0.21726288117213965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6968664339","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052548844,0.00022777659,0.9039702,0.0003426765,0.000055374265,0.00043371756,0.0006979635,0.037781134,0.0039422815],"genre_scores_gemma":[0.14130211,0.00017744653,0.85085404,0.00013584954,0.000031545398,0.00019948184,0.0027198931,0.00088355364,0.0036960815],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985868,0.00026319493,0.00023213892,0.00030412636,0.0005494809,0.00006427131],"domain_scores_gemma":[0.9964037,0.0012742718,0.0002475354,0.00086498185,0.0010266479,0.00018290173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026307567,0.00053722976,0.0006941675,0.0047965734,0.000932604,0.0022216113,0.001082837,0.00074411015,0.0021220087],"category_scores_gemma":[0.0057079475,0.0005403908,0.0011383774,0.0028100235,0.0006905905,0.004148753,0.0028958844,0.0009441876,0.001339928],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000996861,0.0007799056,0.025262289,0.0008308566,0.00032649128,0.0010071037,0.001865995,0.015762134,0.08368396,0.03339342,0.013137569,0.82295346],"study_design_scores_gemma":[0.0002075355,0.00039442818,0.009418798,0.00020701838,0.0002947371,0.0017363499,0.00071081624,0.7018196,0.14440073,0.03321642,0.10736847,0.00022511315],"about_ca_topic_score_codex":0.0022391172,"about_ca_topic_score_gemma":0.0029622174,"teacher_disagreement_score":0.0047965734,"about_ca_system_score_codex":0.000686125,"about_ca_system_score_gemma":0.0012995234,"threshold_uncertainty_score":0.013912916},"labels":[],"label_agreement":null},{"id":"W6986709931","doi":"","title":"Raison et sentiment : nationalisme et antinationalisme dans le Québec des années 1935-1939","year":2001,"lang":"fr","type":"other","venue":"Papyrus : Institutional Repository (Université de Montréal)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Subject (documents); Context (archaeology); Perspective (graphical); Government (linguistics)","score_opus":0.009681276199352716,"score_gpt":0.2072358239816379,"score_spread":0.19755454778228518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6986709931","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.898984,0.0066582854,0.00057531975,0.008638184,0.00016554723,0.00003489642,0.008280876,0.00003375716,0.07662913],"genre_scores_gemma":[0.95066994,0.001606902,0.00023737895,0.00017506449,0.000036612877,0.000020567284,0.00096353167,0.000015797903,0.046274167],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.99954,0.00007619509,0.000017726172,0.0000553107,0.00014206323,0.00016881774],"domain_scores_gemma":[0.9989986,0.00021426496,0.00017003683,0.000026941116,0.00045115233,0.00013899458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000711824,0.00027031856,0.00028686007,0.001737323,0.0039512687,0.0022378934,0.0005377936,0.0004266821,0.0085913725],"category_scores_gemma":[0.0017877596,0.00016373426,0.00019449928,0.0059024254,0.0019228834,0.0010139347,0.0006463025,0.001459856,0.00024937157],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005031044,0.000142282,0.41842452,0.0005379908,0.0002661327,0.0016805625,0.24744889,0.0019072449,0.0023713463,0.14351338,0.08542488,0.097779624],"study_design_scores_gemma":[0.000013409287,0.000020769146,0.81035125,0.00015785401,0.00002993582,0.00010035916,0.038979087,0.00039778955,0.00045332828,0.0010141776,0.14844368,0.000038372465],"about_ca_topic_score_codex":0.9941316,"about_ca_topic_score_gemma":0.99685067,"teacher_disagreement_score":0.040986873,"about_ca_system_score_codex":0.040986873,"about_ca_system_score_gemma":0.022977402,"threshold_uncertainty_score":0.297382},"labels":[],"label_agreement":null},{"id":"W7027413287","doi":"","title":"Chlorine dioxide as a potable water disinfectant : application, residuals, and by-products monitoring","year":2010,"lang":"en","type":"dissertation","venue":"Mspace (University of Manitoba)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Chlorine dioxide; Potassium permanganate; Chlorite; Chlorine; Diethanolamine; Triethanolamine; Chloramine; Permanganate; Trihalomethane; Disinfectant","score_opus":0.006066921726660929,"score_gpt":0.21328831197699288,"score_spread":0.20722139025033195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7027413287","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9686307,0.0074554277,0.018365204,0.0001977039,0.000036997983,0.00019261955,0.00067834,0.00025724486,0.0041859266],"genre_scores_gemma":[0.9725061,0.004126559,0.014016785,0.00013966914,0.000019945572,0.0000926077,0.00092923804,0.00005066442,0.008118468],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99923503,0.00009253352,0.000035436846,0.00015420027,0.000428522,0.00005444675],"domain_scores_gemma":[0.9996619,0.000057565387,0.000066516164,0.000032301603,0.0001662653,0.00001553583],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007176457,0.00041363222,0.00025800665,0.00041439396,0.00025787484,0.0005591556,0.00065449515,0.0007182142,0.001176167],"category_scores_gemma":[0.00050833746,0.00016439757,0.00020227404,0.00038427743,0.0002931402,0.00027126246,0.00024811653,0.0003076149,0.00046935127],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010431489,0.00004054875,0.0027239504,0.00024108445,0.0000101683445,0.00009551388,0.00003145913,0.0003503021,0.98542744,0.00005888903,0.00011152178,0.010804733],"study_design_scores_gemma":[0.0000033576862,0.0003739192,0.0051962254,0.000007171683,0.00001059902,0.00014265417,0.000026570315,0.0006916071,0.99162877,0.000016754331,0.0018958277,0.0000065943245],"about_ca_topic_score_codex":0.006964274,"about_ca_topic_score_gemma":0.012398051,"teacher_disagreement_score":0.006964274,"about_ca_system_score_codex":0.0008806239,"about_ca_system_score_gemma":0.0007416967,"threshold_uncertainty_score":0.01384747},"labels":[],"label_agreement":null},{"id":"W7036213512","doi":"","title":"Best available scientific information on the effects of deposition of heavy metals from long-range atmospheric transport","year":2006,"lang":"en","type":"other","venue":"NERC Open Research Archive (Natural Environment Research Council)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Deposition (geology); Heavy metals; Air pollution; Atmosphere (unit); Pollution","score_opus":0.0505802275379201,"score_gpt":0.30473740211399536,"score_spread":0.25415717457607523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036213512","genre_codex":"dataset","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008577372,0.3653303,0.0026445878,0.016760727,0.0025047269,0.00066516007,0.37686154,0.0007698729,0.22588566],"genre_scores_gemma":[0.10014743,0.56498826,0.01139163,0.0075222813,0.0015037239,0.0007913424,0.20821618,0.00040146773,0.10503773],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.997001,0.00023874054,0.0003524684,0.00019968017,0.001938604,0.00026941905],"domain_scores_gemma":[0.9874374,0.0011692389,0.000982353,0.00033379399,0.009420997,0.000656234],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025935574,0.0012587149,0.0013650319,0.009963061,0.0020671966,0.0020702395,0.0026388553,0.0010701714,0.03751541],"category_scores_gemma":[0.008388358,0.00050397794,0.0014190408,0.018363345,0.00073780393,0.0011071555,0.0010338043,0.0009693261,0.0074571483],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027933365,0.00007608228,0.018149504,0.023971919,0.00037818178,0.00034015576,0.00046672413,0.0006239541,0.00094082544,0.0021629813,0.584666,0.36794427],"study_design_scores_gemma":[0.000041303625,0.000036829864,0.07287583,0.009418623,0.0004373509,0.000283213,0.0003425187,0.00016550923,0.0006929895,0.00067399506,0.9149762,0.000055658817],"about_ca_topic_score_codex":0.92921054,"about_ca_topic_score_gemma":0.9543193,"teacher_disagreement_score":0.92921054,"about_ca_system_score_codex":0.016354281,"about_ca_system_score_gemma":0.059717853,"threshold_uncertainty_score":0.14241266},"labels":[],"label_agreement":null},{"id":"W7096651494","doi":"","title":"Published by Canadian Center of Science and Education 191 Testing the Accuracy of Text Deconstruction Using PTree Tool","year":2015,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Correctness; Parsing; Deconstruction (building); Java; Relation (database); Flexibility (engineering); Parse tree; Tree (set theory)","score_opus":0.03378848661564744,"score_gpt":0.29689182451678287,"score_spread":0.26310333790113544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096651494","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052603963,0.0010934193,0.016578851,0.00792292,0.0017076533,0.00036157793,0.03232878,0.0058378153,0.9289085],"genre_scores_gemma":[0.027300721,0.0022994122,0.033397228,0.0010383286,0.000083898856,0.00019453658,0.019930033,0.0019223505,0.91383356],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967811,0.00019126927,0.00014549124,0.0004280089,0.002164182,0.00028992948],"domain_scores_gemma":[0.983303,0.001116228,0.00023680484,0.0016527269,0.012601571,0.001089709],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0027446472,0.0006145566,0.00090154854,0.0068994,0.0046980553,0.008269075,0.001780665,0.001313518,0.37134823],"category_scores_gemma":[0.009978891,0.0006427091,0.00051632174,0.009264997,0.0020123424,0.0027634571,0.0024031845,0.0014419971,0.13190697],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008906743,0.00008661656,0.0021693588,0.00025954566,0.000011330446,0.00023296296,0.0005230521,0.0005743637,0.001931339,0.052311514,0.65606165,0.2857493],"study_design_scores_gemma":[0.000009498743,0.0000090020685,0.0017657556,0.0000747226,0.0000042955908,0.00007199035,0.00029427028,0.0006992585,0.00066196546,0.0012090346,0.99517953,0.000020708538],"about_ca_topic_score_codex":0.72418976,"about_ca_topic_score_gemma":0.8224137,"teacher_disagreement_score":0.72418976,"about_ca_system_score_codex":0.017941525,"about_ca_system_score_gemma":0.062501036,"threshold_uncertainty_score":0.89669544},"labels":[],"label_agreement":null},{"id":"W7098708152","doi":"","title":"REDUCED POWER AT TAKE-OFF AND COLLISION WITH TERRAIN","year":2004,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Terrain; Collision; Function (biology); Power (physics); Fault (geology); Aviation","score_opus":0.00579598134202386,"score_gpt":0.2401106088972919,"score_spread":0.23431462755526805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7098708152","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9527349,0.0005210401,0.009822187,0.00079412886,0.0001998413,0.00007034773,0.00047916174,0.00029069197,0.035087597],"genre_scores_gemma":[0.9939936,0.00012362096,0.00078963005,0.00006214666,0.000028649785,0.000007812617,0.0001300596,0.000021724492,0.004842695],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9994779,0.000024279529,0.000009743933,0.000034916204,0.00032347147,0.00012959835],"domain_scores_gemma":[0.9994086,0.00013108591,0.00013193135,0.0000774506,0.0001496974,0.00010133622],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001514687,0.0002509412,0.00027222544,0.0010697616,0.0011181692,0.0005381648,0.0004988976,0.00053466024,0.007336653],"category_scores_gemma":[0.0017107014,0.00012720328,0.00036813875,0.00076557713,0.0006117247,0.00030433116,0.00065108546,0.0005470104,0.0008783876],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0040355385,0.00054905855,0.3352783,0.000463171,0.00028832126,0.11701726,0.005927062,0.014169173,0.08283422,0.0135315815,0.032181952,0.3937243],"study_design_scores_gemma":[0.0001560136,0.001380113,0.8334736,0.00010382799,0.00012446985,0.06861595,0.0039077643,0.013071265,0.026416378,0.013785297,0.03887575,0.000089611065],"about_ca_topic_score_codex":0.011146458,"about_ca_topic_score_gemma":0.019300275,"teacher_disagreement_score":0.011146458,"about_ca_system_score_codex":0.00050583517,"about_ca_system_score_gemma":0.0003936482,"threshold_uncertainty_score":0.024543524},"labels":[],"label_agreement":null},{"id":"W7100179867","doi":"","title":"Text Processing","year":2016,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Relevance (law); Key (lock); Focus (optics); Ranking (information retrieval); Work (physics); Quarter (Canadian coin); Recommender system","score_opus":0.010890496970892501,"score_gpt":0.27193809495669846,"score_spread":0.26104759798580596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7100179867","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01278781,0.0106723495,0.21332766,0.008707079,0.009604918,0.0031638034,0.19275743,0.047585703,0.5013932],"genre_scores_gemma":[0.07350787,0.0071364013,0.17465045,0.0043702144,0.002741034,0.0018153862,0.21454985,0.005987214,0.5152415],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981737,0.00024585117,0.00024294914,0.0005018342,0.0007032323,0.00013239677],"domain_scores_gemma":[0.9973756,0.00047568872,0.00015332334,0.00062270963,0.0012817184,0.00009089602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012794945,0.0012704966,0.00089820847,0.0044579264,0.00095513655,0.0046497555,0.0012425913,0.0010360571,0.2014431],"category_scores_gemma":[0.005278166,0.00033561446,0.0012895291,0.004884692,0.00043461288,0.0027065342,0.0017761127,0.0011409813,0.2548973],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022536912,0.00006493225,0.000801776,0.001010492,0.00007760012,0.0002197954,0.00016430233,0.0009058393,0.0054551866,0.008600252,0.3162438,0.6662307],"study_design_scores_gemma":[0.000043995286,0.000068929396,0.0019355896,0.00021468138,0.000054593413,0.00027968027,0.00028059797,0.004545239,0.0070124846,0.009564591,0.9759591,0.000040493498],"about_ca_topic_score_codex":0.0023925924,"about_ca_topic_score_gemma":0.002170439,"teacher_disagreement_score":0.2014431,"about_ca_system_score_codex":0.00070134946,"about_ca_system_score_gemma":0.001552873,"threshold_uncertainty_score":0.6738943},"labels":[],"label_agreement":null},{"id":"W7100658452","doi":"","title":"Adaptation of a Keyphrase Extractor for Japanese","year":2007,"lang":"en","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Extractor; Software; Adaptation (eye); Selection (genetic algorithm); Information extraction; Matching (statistics)","score_opus":0.026469365737060692,"score_gpt":0.315559452077982,"score_spread":0.28909008634092126,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7100658452","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015311352,0.0006641077,0.89246446,0.00023632267,0.00050705974,0.00059160485,0.0030961477,0.07999022,0.0071387994],"genre_scores_gemma":[0.03578376,0.00053694943,0.9299646,0.00019862525,0.00014646881,0.00042748917,0.007347341,0.010235557,0.015359157],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99844295,0.00013453017,0.00025664922,0.0005165037,0.0005140217,0.00013537753],"domain_scores_gemma":[0.9962829,0.00071716995,0.00016947629,0.00089501153,0.0017493695,0.00018602256],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015397934,0.0020131012,0.0015657784,0.0029751428,0.0012453285,0.0022404045,0.0015054725,0.000971093,0.020082893],"category_scores_gemma":[0.0061913175,0.0012611414,0.0015834821,0.0030324257,0.00058916333,0.0038232522,0.0019301095,0.0018236466,0.015501048],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001068401,0.00014718021,0.002331189,0.0011893668,0.00021279987,0.00085837144,0.0013640892,0.0017238227,0.15131837,0.003986088,0.03187458,0.80392563],"study_design_scores_gemma":[0.00039040105,0.0004898702,0.0110655045,0.00021367257,0.0005136977,0.0032468336,0.00093348726,0.07783695,0.23605159,0.004396379,0.66429454,0.00056714466],"about_ca_topic_score_codex":0.0077208425,"about_ca_topic_score_gemma":0.010448818,"teacher_disagreement_score":0.020082893,"about_ca_system_score_codex":0.00073705794,"about_ca_system_score_gemma":0.0016401394,"threshold_uncertainty_score":0.06718397},"labels":[],"label_agreement":null},{"id":"W7120589744","doi":"","title":"Análise visual de tópicos no Twitter em conexão com debates políticos","year":2017,"lang":"en","type":"dissertation","venue":"LA Referencia (Red Federada de Repositorios Institucionales de Publicaciones Científicas)","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Social media; Government (linguistics); Set (abstract data type); Parliament; Dissemination; Cluster analysis; Topic model; Politics","score_opus":0.019344523219971974,"score_gpt":0.2897597519075543,"score_spread":0.27041522868758233,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7120589744","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91778016,0.0012422899,0.033457622,0.0011050703,0.00026851596,0.00025498267,0.010080081,0.0051655043,0.030645791],"genre_scores_gemma":[0.9577247,0.0006571164,0.028767813,0.000064967106,0.00017381541,0.00014971897,0.004949236,0.0005153208,0.006997363],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996586,0.000053503878,0.0000236748,0.000066409084,0.00013789498,0.000059857415],"domain_scores_gemma":[0.99813974,0.0009143241,0.00020179767,0.00015421466,0.00048100553,0.0001089989],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068736786,0.0004135029,0.00029673352,0.006127431,0.0006499035,0.0018022765,0.00025581758,0.0005026236,0.0045710495],"category_scores_gemma":[0.0027275456,0.00018757765,0.0003685756,0.003612728,0.0003512626,0.0013320189,0.000964628,0.00044392416,0.0010954567],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026607257,0.0002838551,0.1427251,0.0029102252,0.00023691032,0.006514301,0.053303733,0.008169868,0.16840288,0.016433412,0.04485584,0.55350316],"study_design_scores_gemma":[0.00008475378,0.00029746658,0.52686954,0.0008129449,0.000302561,0.0026853164,0.05907709,0.12877584,0.060205884,0.011897386,0.20875978,0.00023141413],"about_ca_topic_score_codex":0.0031800838,"about_ca_topic_score_gemma":0.0048702126,"teacher_disagreement_score":0.006127431,"about_ca_system_score_codex":0.00029208596,"about_ca_system_score_gemma":0.00026985942,"threshold_uncertainty_score":0.015291691},"labels":[],"label_agreement":null},{"id":"W7130592618","doi":"10.1109/fllm67465.2025.11391003","title":"Automated Research Article Classification and Recommendation Using NLP and Machine Learning","year":2025,"lang":"","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Algoma University; York University","funders":"","keywords":"Cosine similarity; Support vector machine; Random forest; Information overload; Complement (music); Statistical classification; Feature extraction; Feature (linguistics)","score_opus":0.1414599886026211,"score_gpt":0.44178727865896866,"score_spread":0.30032729005634756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7130592618","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09867315,0.010428889,0.7166359,0.0057613105,0.0007558769,0.0025406228,0.050487857,0.095398195,0.019318163],"genre_scores_gemma":[0.17944245,0.002092261,0.7555246,0.0005348682,0.0006275048,0.00061491894,0.053009104,0.00047809348,0.0076762317],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956995,0.00082066585,0.000528673,0.0010008179,0.0016928171,0.0002575146],"domain_scores_gemma":[0.9863751,0.0047383765,0.0019593083,0.002080869,0.0043754387,0.00047093048],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030361763,0.0012191306,0.0020168908,0.020033775,0.001365619,0.0028262283,0.0017777737,0.0014341958,0.003306221],"category_scores_gemma":[0.015800972,0.00050026266,0.0013505747,0.012923283,0.00041965468,0.0034535795,0.0010745202,0.0011232134,0.008706959],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028326744,0.0005214092,0.019274581,0.0007529747,0.00020083375,0.00030522142,0.00017230051,0.005831788,0.018187558,0.0022403032,0.058026478,0.89420325],"study_design_scores_gemma":[0.00031155057,0.0004661194,0.030552858,0.00029295322,0.00034093455,0.0012760928,0.0007739754,0.7457172,0.0757379,0.028651278,0.11566013,0.00021897035],"about_ca_topic_score_codex":0.014762586,"about_ca_topic_score_gemma":0.026476547,"teacher_disagreement_score":0.020033775,"about_ca_system_score_codex":0.001328405,"about_ca_system_score_gemma":0.0034161734,"threshold_uncertainty_score":0.02935332},"labels":[],"label_agreement":null},{"id":"W7147479603","doi":"10.1109/icaft66710.2025.11452853","title":"Unsupervised Key-Value Pair selection on Enterprise Documents through Generative type Transformer Framework","year":2025,"lang":"","type":"article","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Generative grammar; Transformer; Unsupervised learning; Automation; Generative model; Generalization","score_opus":0.014170597863199184,"score_gpt":0.3244799686035344,"score_spread":0.3103093707403352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7147479603","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017845483,0.00016378504,0.9766912,0.00016788766,0.00002289732,0.00012751944,0.0003135726,0.0025821358,0.0020855844],"genre_scores_gemma":[0.41147882,0.00031719782,0.5779653,0.00017603299,0.00005802345,0.0001857758,0.001852464,0.0006369555,0.007329345],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985864,0.0003841064,0.00008693459,0.00040962163,0.0003962172,0.00013678518],"domain_scores_gemma":[0.9977773,0.0011598156,0.00017466268,0.0004388365,0.00035775756,0.000091697926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014632257,0.00060865114,0.0008859683,0.002772232,0.00066686946,0.002086539,0.0018232405,0.0008311056,0.0031953186],"category_scores_gemma":[0.0048494996,0.00045483134,0.0014479783,0.003154363,0.0011179657,0.004272456,0.0018049722,0.0013093076,0.0018995039],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005353802,0.00023765353,0.0061879093,0.00035194133,0.00013323555,0.00090814097,0.0011229304,0.06658363,0.027873129,0.12904891,0.009571845,0.7574453],"study_design_scores_gemma":[0.00004783218,0.00008008008,0.0010980559,0.000035110468,0.000075896125,0.00066666765,0.00025945302,0.85101205,0.035581067,0.09600493,0.015084,0.000054887118],"about_ca_topic_score_codex":0.0028509207,"about_ca_topic_score_gemma":0.0045312396,"teacher_disagreement_score":0.0031953186,"about_ca_system_score_codex":0.0010759004,"about_ca_system_score_gemma":0.0017896335,"threshold_uncertainty_score":0.010689378},"labels":[],"label_agreement":null},{"id":"W81794110","doi":"","title":"Adaptive User Interfaces for Intelligent E-Learning: Issues and Trends","year":2004,"lang":"en","type":"article","venue":"Journal of the Association for Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; University of Waterloo","funders":"","keywords":"Computer science; Human–computer interaction; User interface; The Internet; Context (archaeology); Multimedia; Domain (mathematical analysis); World Wide Web; Process (computing); User interface design; User experience design","score_opus":0.01541633088549207,"score_gpt":0.2851988313401981,"score_spread":0.26978250045470603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W81794110","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009099871,0.8596975,0.036549143,0.051165838,0.0014088028,0.00012793436,0.00009990179,0.000631294,0.04121969],"genre_scores_gemma":[0.11506347,0.75937635,0.08904625,0.014351432,0.006195747,0.0004713548,0.00029975537,0.0003309435,0.01486471],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9962829,0.0010817428,0.00046484583,0.0005128504,0.001485148,0.00017257205],"domain_scores_gemma":[0.9754362,0.016904658,0.0006747269,0.00075890747,0.005511897,0.00071351056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00918144,0.00072154973,0.0011287045,0.0030803853,0.00082596834,0.008028675,0.0024715567,0.0061959545,0.00541063],"category_scores_gemma":[0.013142542,0.0006115851,0.0004950131,0.0060206275,0.0032299212,0.017524099,0.0019655097,0.0053312317,0.0024912567],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014838323,0.00026016598,0.002421733,0.0040692263,0.000036310543,0.00010637116,0.0017181332,0.0008806461,0.0020009817,0.116637215,0.019708732,0.8520121],"study_design_scores_gemma":[0.00006268818,0.0005440139,0.0050738025,0.0064467294,0.000082731036,0.0014432975,0.0046106526,0.010231003,0.0031789616,0.10085242,0.8673018,0.00017188565],"about_ca_topic_score_codex":0.0011836507,"about_ca_topic_score_gemma":0.0011539552,"teacher_disagreement_score":0.00918144,"about_ca_system_score_codex":0.0018193403,"about_ca_system_score_gemma":0.0015859418,"threshold_uncertainty_score":0.048556685},"labels":[],"label_agreement":null},{"id":"W86593375","doi":"10.4018/978-1-60566-058-5.ch020","title":"Theories of Meaning in Schema Matching","year":2009,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Schema matching; Schema (genetic algorithms); Matching (statistics); Conceptual schema; Computer science; Epistemology; Schema migration; Information retrieval; Database schema; Psychology; Data mining; Data integration; Semi-structured model; Social psychology; Gender schema theory; Mathematics; Philosophy; Database design","score_opus":0.012164443046926097,"score_gpt":0.2655339347594541,"score_spread":0.253369491712528,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W86593375","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009940655,0.02676463,0.6797311,0.013720503,0.00094137853,0.00026535953,0.0003643162,0.0004913461,0.26778075],"genre_scores_gemma":[0.30285054,0.033974703,0.60448706,0.0046762074,0.0014344192,0.0011445942,0.0014949606,0.0006615974,0.049275912],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9958234,0.0021520338,0.00031150726,0.00063090574,0.00090366486,0.00017852636],"domain_scores_gemma":[0.99356234,0.0047798716,0.00026544638,0.00085468654,0.0004001872,0.00013745774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005435693,0.0010628195,0.00076046184,0.005114724,0.0023771701,0.009054049,0.0027066283,0.0028156089,0.011051441],"category_scores_gemma":[0.012628885,0.00078953826,0.0017150792,0.007639238,0.017652847,0.027405676,0.003981873,0.0043362756,0.0031554848],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000004539185,0.000006633081,0.00008230332,0.00009530096,0.0000067100796,0.00003361606,0.0012279953,0.00028923893,0.000068309266,0.98043,0.001927877,0.01582742],"study_design_scores_gemma":[0.000004771746,0.0000047863123,0.000097776116,0.000113983755,0.000006882279,0.00011770185,0.00040624043,0.0011110376,0.00014206355,0.96526295,0.032724362,0.000007389561],"about_ca_topic_score_codex":0.0014556311,"about_ca_topic_score_gemma":0.0010591456,"teacher_disagreement_score":0.011051441,"about_ca_system_score_codex":0.003941406,"about_ca_system_score_gemma":0.0021563494,"threshold_uncertainty_score":0.036970794},"labels":[],"label_agreement":null},{"id":"W86784885","doi":"10.48009/2_iis_2005_296-302","title":"DO YOU HEAR WHAT I HEAR? ADVANCES IN WEB-BASED PERCEPTUAL TESTING AND TRAINING","year":2005,"lang":"en","type":"article","venue":"Issues in Information Systems","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Perception; Training (meteorology); Psychology; Computer science; Cognitive psychology; Neuroscience; Geography","score_opus":0.027014595235242994,"score_gpt":0.3105845741142363,"score_spread":0.28356997887899327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W86784885","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15277888,0.078000724,0.5657771,0.014529573,0.0016635797,0.00084552827,0.0006091459,0.00502654,0.18076892],"genre_scores_gemma":[0.45866707,0.041305,0.46719304,0.0023912203,0.0020810342,0.000787462,0.0005833677,0.00077699113,0.026214825],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99401516,0.0030202596,0.00023158667,0.0005712224,0.0020026537,0.00015911969],"domain_scores_gemma":[0.9388118,0.049887568,0.001736443,0.0029933539,0.0052523324,0.001318572],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008942835,0.0005872054,0.0005407945,0.002504254,0.0002654833,0.002833358,0.002396505,0.0014175504,0.010055273],"category_scores_gemma":[0.026246268,0.00036352442,0.00037807666,0.001494828,0.0019276902,0.004145135,0.0013528541,0.001219594,0.003151303],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001334437,0.00044968538,0.0030210107,0.00047181218,0.000014415356,0.000046693018,0.00078918255,0.0006348654,0.0022141903,0.004900452,0.0034582443,0.98386604],"study_design_scores_gemma":[0.0003408553,0.0033753708,0.090957195,0.004106815,0.00020458079,0.005832446,0.0055044373,0.07361352,0.040134184,0.06803082,0.7072235,0.00067634584],"about_ca_topic_score_codex":0.0031149238,"about_ca_topic_score_gemma":0.002450482,"teacher_disagreement_score":0.010055273,"about_ca_system_score_codex":0.0008858323,"about_ca_system_score_gemma":0.0010805622,"threshold_uncertainty_score":0.047294736},"labels":[],"label_agreement":null},{"id":"W923756419","doi":"","title":"University of Waterloo at TREC 2014 Contextual Suggestion: Experiments with suggestion clustering","year":2014,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Point of interest; Task (project management); Point (geometry); Similarity (geometry); Cluster analysis; Information retrieval; World Wide Web; Special Interest Group; Artificial intelligence; Mathematics","score_opus":0.01781061543086193,"score_gpt":0.24038491729818853,"score_spread":0.2225743018673266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W923756419","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8498437,0.009122601,0.018169438,0.0052189543,0.0015937181,0.0035681578,0.027655846,0.0283323,0.05649526],"genre_scores_gemma":[0.8015736,0.0020512326,0.09807819,0.0016253596,0.0005813411,0.0017092366,0.06631204,0.0015769263,0.026492052],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98984677,0.00541449,0.00058424263,0.0015490097,0.0021276614,0.0004779139],"domain_scores_gemma":[0.96407396,0.023628952,0.00089454214,0.0039149947,0.0053250124,0.0021624872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008740128,0.0016074046,0.0018304938,0.0017439008,0.003640552,0.0020924788,0.0030729554,0.0025182085,0.012655079],"category_scores_gemma":[0.0350465,0.00077034306,0.0007640295,0.0037198332,0.0011076708,0.0038697512,0.0016076466,0.0026187159,0.005004578],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010977005,0.014391465,0.019856961,0.003895808,0.00093206123,0.0006633989,0.002856076,0.03945339,0.019371066,0.00246331,0.48171222,0.40342727],"study_design_scores_gemma":[0.009393843,0.008971638,0.096434325,0.000846573,0.0011508681,0.000775324,0.0063138525,0.6733085,0.04019113,0.006049659,0.15564002,0.00092426053],"about_ca_topic_score_codex":0.18286334,"about_ca_topic_score_gemma":0.20865199,"teacher_disagreement_score":0.18286334,"about_ca_system_score_codex":0.004432694,"about_ca_system_score_gemma":0.0045464463,"threshold_uncertainty_score":0.36359793},"labels":[],"label_agreement":null},{"id":"W9417692","doi":"10.1007/978-3-319-06483-3_22","title":"Combining Textual Pre-game Reports and Statistical Data for Predicting Success in the National Hockey League","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Bigram; Trigram; Computer science; League; Classifier (UML); Artificial intelligence; Machine learning; Natural language processing; Data mining","score_opus":0.03321151851607614,"score_gpt":0.3239491014298272,"score_spread":0.29073758291375107,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W9417692","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9690587,0.0006994016,0.013041651,0.0003596412,0.000131401,0.000093882816,0.010637841,0.0006247747,0.0053527546],"genre_scores_gemma":[0.9671844,0.00032065832,0.010824272,0.000049708946,0.00015814476,0.00011554371,0.018088501,0.000048649264,0.0032101278],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991347,0.00030628993,0.000082230705,0.00013520248,0.0002603925,0.00008124665],"domain_scores_gemma":[0.98725605,0.009670879,0.001065535,0.0003612985,0.0011886823,0.00045757208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013616048,0.00077244034,0.00033468133,0.0036359762,0.0002792051,0.0015275293,0.0007243114,0.0007724583,0.0025527547],"category_scores_gemma":[0.009572527,0.00028946152,0.00038502255,0.0025202048,0.00021601822,0.0012225675,0.0006029299,0.0007856715,0.00273533],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00087379396,0.0012719412,0.74048865,0.00026903278,0.00026045166,0.00026748996,0.00039598165,0.008978041,0.0040780585,0.00037606468,0.011660372,0.23108022],"study_design_scores_gemma":[0.000059417765,0.0012648078,0.72097796,0.0001975482,0.0003344048,0.00032348227,0.0019565437,0.26016682,0.0067984895,0.0020297635,0.00576845,0.00012237327],"about_ca_topic_score_codex":0.0074396953,"about_ca_topic_score_gemma":0.018882435,"teacher_disagreement_score":0.0074396953,"about_ca_system_score_codex":0.00031037309,"about_ca_system_score_gemma":0.00044746575,"threshold_uncertainty_score":0.0147928},"labels":[],"label_agreement":null},{"id":"W98088564","doi":"10.29173/cais472","title":"Team Co-occurence in Internet Search Engine Queries: An Analysis of the Excite Data Set","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Set (abstract data type); Computer science; Search engine; Term (time); Zipf's law; The Internet; Information retrieval; Data set; Data mining; Distribution (mathematics); Statistics; World Wide Web; Mathematics; Artificial intelligence; Physics","score_opus":0.04172340190283944,"score_gpt":0.30678274582578985,"score_spread":0.2650593439229504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W98088564","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99160355,0.00006492051,0.00027450945,0.00006509168,0.0000021382423,0.000014784086,0.007580078,0.000030503084,0.00036437952],"genre_scores_gemma":[0.9605513,0.00009841603,0.0009959236,0.000028767192,0.000013107552,0.000058164107,0.037516955,0.000027770555,0.00070954615],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99789196,0.0006687709,0.0002876721,0.0003731946,0.0005984325,0.00017995486],"domain_scores_gemma":[0.97826844,0.015144342,0.0027792586,0.0015887192,0.0014993262,0.0007198996],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002361218,0.00026538942,0.00078683294,0.0063130935,0.00042975752,0.0010368283,0.00051904406,0.0006200517,0.0015618644],"category_scores_gemma":[0.014170168,0.00019230138,0.00076894846,0.00709183,0.00049659534,0.0011610908,0.0012297367,0.00047709583,0.000978895],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000759663,0.000268445,0.9686582,0.00025689154,0.00034924923,0.00049640785,0.0014475508,0.0036861994,0.0027580569,0.00030891056,0.0029087178,0.018101618],"study_design_scores_gemma":[0.000026735783,0.00018229813,0.9819516,0.000016490534,0.000059510385,0.00050574733,0.0018494433,0.01211215,0.001075757,0.00014731786,0.0020389785,0.000034012442],"about_ca_topic_score_codex":0.014217285,"about_ca_topic_score_gemma":0.02070296,"teacher_disagreement_score":0.014217285,"about_ca_system_score_codex":0.0007112463,"about_ca_system_score_gemma":0.0005337089,"threshold_uncertainty_score":0.028269053},"labels":[],"label_agreement":null}]}