{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":41,"total_is_capped":false,"direct_labels_cover":1,"predictions_cover":41,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"f029b2e5f93c","filters":{"venue":"Data & Knowledge Engineering"}},"results":[{"id":"W2043640532","doi":"10.1016/j.datak.2004.12.009","title":"Complexity and clarity in conceptual modeling: Comparison of mandatory and optional properties","year":2005,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":239,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia; Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"CLARITY; Computer science; Grammar; Cognitive psychology; Natural language processing; Rule-based machine translation; Domain (mathematical analysis); Domain model; Linguistics; Psychology; Artificial intelligence; Domain knowledge; Mathematics","authors":[{"name":"Andrew Gemino","is_ca":true},{"name":"Yair Wand","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2068040600089384,"gpt":0.3213727864468606,"spread":0.1145687264379222,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03382764,0.0006411945,0.001116196,0.005034317,0.001970216,0.008770312,0.002337766,0.002362881,0.00347339],"category_scores_gemma":[0.1901595,0.001270463,0.0028839,0.002848749,0.006286419,0.03101921,0.005235642,0.00314922,0.0003575285],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002785674,"about_ca_system_score_gemma":0.002726292,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001938796,"about_ca_topic_score_gemma":0.001848945,"domain_scores_codex":[0.9672541,0.01738322,0.002984583,0.001404837,0.009811774,0.001161466],"domain_scores_gemma":[0.6466206,0.2916571,0.01680204,0.02600557,0.01604778,0.002866827],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008382776,0.0001710676,0.01718072,0.0007463022,0.0002386593,0.0002174487,0.006827567,0.0227567,0.003650745,0.8801969,0.001837136,0.06533858],"study_design_scores_gemma":[0.0001074884,0.0002214628,0.008904579,0.0004259837,0.0003629747,0.0003620112,0.00187039,0.1559051,0.005998082,0.8202013,0.005505945,0.0001347647],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2797855,0.001792548,0.691824,0.003411695,0.0001592336,0.0003002284,0.0005127535,0.0005448741,0.02166927],"genre_scores_gemma":[0.9130039,0.0005467187,0.08490016,0.0001376297,0.00009648815,0.0001423402,0.000444105,0.0001891162,0.0005396481],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03382764,"threshold_uncertainty_score":0.1788998,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2145399176","doi":"10.1016/j.datak.2006.06.001","title":"Combined mining of Web server logs and web contents for classifying user navigation patterns and predicting users’ future requests","year":2006,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":205,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Dalhousie University","funders":"","keywords":"Web mining; Computer science; Web page; Web server; Web navigation; Personalization; Static web page; World Wide Web; Web log analysis software; Information retrieval; Data Web; Web API; Database; Data mining; The Internet","authors":[{"name":"Haibin Liu","is_ca":true},{"name":"Vlado Kešelj","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03013581955046483,"gpt":0.2583486271505941,"spread":0.2282128076001293,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000991649,0.0007931046,0.001525321,0.006811072,0.0005696947,0.001380512,0.0007824204,0.0009524498,0.000631808],"category_scores_gemma":[0.004763849,0.0004073525,0.001249805,0.005124612,0.0001676625,0.00150751,0.000465096,0.0008033989,0.0009312504],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003358589,"about_ca_system_score_gemma":0.0009480842,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009285218,"about_ca_topic_score_gemma":0.02791598,"domain_scores_codex":[0.998714,0.000303914,0.0001513771,0.0002565767,0.0004828214,0.00009144963],"domain_scores_gemma":[0.9951105,0.002461871,0.0004445473,0.0006334841,0.001120576,0.0002291227],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007183869,0.002888974,0.3136686,0.0005949721,0.001134574,0.0003856215,0.0006213582,0.01536555,0.04680341,0.0007321088,0.005879508,0.6112069],"study_design_scores_gemma":[0.0000555029,0.0006317848,0.1582406,0.00007088768,0.0007350749,0.0007850746,0.0004334004,0.8120016,0.02095196,0.002407164,0.003595738,0.00009131998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7530171,0.001509394,0.2298942,0.0004086847,0.000121379,0.0003148341,0.006520971,0.004970246,0.003243297],"genre_scores_gemma":[0.8739886,0.0004681575,0.1166761,0.00008075227,0.00009493824,0.0001591779,0.006468415,0.00008656913,0.001977284],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009285218,"threshold_uncertainty_score":0.01846236,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1984778493","doi":"10.1016/j.datak.2010.01.005","title":"The consistency extractor system: Answer set programs for consistent query answering in databases","year":2010,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Logic, Reasoning, and Knowledge","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Datalog; Answer set programming; Database theory; Consistency (knowledge bases); Extractor; Programming language; Relational database; Deductive database; Set (abstract data type); Information retrieval; Database; Artificial intelligence","authors":[{"name":"Mónica Caniupán","is_ca":false},{"name":"Leopoldo Bertossi","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05562206053334146,"gpt":0.2879527423906523,"spread":0.2323306818573109,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01001042,0.001363872,0.002285329,0.002463933,0.00120717,0.004572512,0.005496748,0.002741568,0.008407577],"category_scores_gemma":[0.0319539,0.001920828,0.002167032,0.002161601,0.002805467,0.009835663,0.00582619,0.004674274,0.002266024],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00106504,"about_ca_system_score_gemma":0.003533918,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003138971,"about_ca_topic_score_gemma":0.00366204,"domain_scores_codex":[0.9923626,0.002457143,0.0007664679,0.001032598,0.002982114,0.0003990468],"domain_scores_gemma":[0.9865016,0.009328169,0.000618442,0.002097421,0.001243286,0.0002110669],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001690126,0.0006278694,0.004383624,0.001529995,0.000434701,0.0005523483,0.001485287,0.04113708,0.01722122,0.2185422,0.06872012,0.6436754],"study_design_scores_gemma":[0.0007000509,0.0002733127,0.0009731199,0.0002771535,0.0004897598,0.0004224362,0.0003501801,0.6052988,0.05433553,0.2889946,0.04770012,0.0001849462],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00322614,0.0001748159,0.9799968,0.0004584052,0.00003037081,0.0001776001,0.0006613549,0.01459269,0.0006819292],"genre_scores_gemma":[0.07574505,0.0003535601,0.9146646,0.0006674454,0.0001371867,0.0004989433,0.002857285,0.002810688,0.00226536],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01001042,"threshold_uncertainty_score":0.05294079,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1966678916","doi":"10.1016/j.datak.2009.08.006","title":"Sorting improves word-aligned bitmap indexes","year":2009,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Parallel Computing and Optimization Techniques","field":"Computer Science","cited_by":74,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick; Université du Québec à Montréal","funders":"","keywords":"Bitmap; Lexicographical order; Sorting; Sorting algorithm; sort; Aggregate (composite); Table (database); Construct (python library); Simple (philosophy)","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.02252072736554118,"gpt":0.2728181693191338,"spread":0.2502974419535927,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001134623,0.001130028,0.001713889,0.003089619,0.001407459,0.003070682,0.002172641,0.001368853,0.01399918],"category_scores_gemma":[0.009863682,0.0006732717,0.000763428,0.009103471,0.000631211,0.006240766,0.001983581,0.0009546016,0.006433142],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007840415,"about_ca_system_score_gemma":0.002762107,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003399842,"about_ca_topic_score_gemma":0.007168607,"domain_scores_codex":[0.9982309,0.0003093193,0.0002365929,0.0002793961,0.0007070568,0.000236785],"domain_scores_gemma":[0.9940929,0.001752675,0.0003503671,0.001840234,0.001761109,0.0002026982],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002466485,0.0007598043,0.003528539,0.0006320882,0.0001419966,0.0002928868,0.0004113984,0.03297233,0.03368431,0.03408295,0.0460066,0.8450206],"study_design_scores_gemma":[0.0008349061,0.001635761,0.003453462,0.0002154061,0.0003902819,0.0008310204,0.001057229,0.6145759,0.1306766,0.1828954,0.06319257,0.0002413606],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3617039,0.005147773,0.5586286,0.001259223,0.002498189,0.0004051901,0.00359108,0.03648591,0.03028017],"genre_scores_gemma":[0.3816653,0.001495872,0.5868776,0.0006165075,0.0004595474,0.0002060875,0.008856897,0.002851871,0.01697033],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01399918,"threshold_uncertainty_score":0.04683197,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2144326136","doi":"10.1016/j.datak.2008.12.001","title":"Privacy-preserving data publishing for cluster analysis","year":2008,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":71,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ontario Institute of Technology; Simon Fraser University; Concordia University","funders":"","keywords":"Masking (illustration); Data publishing; Computer science; Cluster (spacecraft); ENCODE; Data mining; Process (computing); Publishing; Focus (optics); Data anonymization; Class (philosophy); Data quality; Information retrieval; Information privacy; Artificial intelligence; Computer security; Engineering","authors":[{"name":"Benjamin C. M. Fung","is_ca":true},{"name":"Ke Wang","is_ca":true},{"name":"Lingyu Wang","is_ca":true},{"name":"Patrick C. K. Hung","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1037152150446969,"gpt":0.306164189747526,"spread":0.2024489747028291,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006926941,0.0005263381,0.001854571,0.001792489,0.002728708,0.005056927,0.003076259,0.001532212,0.002324322],"category_scores_gemma":[0.03288099,0.0008451137,0.00159527,0.005332273,0.002297281,0.007467228,0.005082353,0.003182583,0.001305502],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001503709,"about_ca_system_score_gemma":0.003811897,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001013557,"about_ca_topic_score_gemma":0.001051106,"domain_scores_codex":[0.9911855,0.002892425,0.0008234037,0.001312297,0.003109221,0.0006770661],"domain_scores_gemma":[0.9556216,0.01076213,0.001690366,0.02847267,0.002801351,0.0006518385],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001338456,0.0003639558,0.006240323,0.000412576,0.0003559277,0.0005107038,0.001313812,0.1009111,0.01305623,0.5428522,0.01780894,0.3148358],"study_design_scores_gemma":[0.00008015711,0.0001250941,0.0007616991,0.00005221397,0.0001186501,0.0008378302,0.0003644412,0.338026,0.03172371,0.6112437,0.01661352,0.00005301265],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01765741,0.0004554803,0.976899,0.001134892,0.0001432694,0.0001020426,0.0006301852,0.0008840122,0.002093802],"genre_scores_gemma":[0.6327802,0.0009427636,0.3554146,0.0005749061,0.000479353,0.0003063797,0.001963372,0.0003865457,0.007151898],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006926941,"threshold_uncertainty_score":0.03663361,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4394987794","doi":"10.1016/j.datak.2024.102306","title":"Effective text classification using BERT, MTM LSTM, and DT","year":2024,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Artificial Intelligence in Medicine (Canada); University of Calgary","funders":"","keywords":"Security token; Computer science; Artificial intelligence; Encoder; Binary classification; Transformer; Recall; Deep learning; Long short term memory; Natural language processing; Artificial neural network; Machine learning; Recurrent neural network; Support vector machine","authors":[{"name":"Saman Jamshidi","is_ca":false},{"name":"Mahin Mohammadi","is_ca":false},{"name":"Saeed Bagheri","is_ca":false},{"name":"Hamid Esmaeili Najafabadi","is_ca":true},{"name":"Alireza Rezvanian","is_ca":true},{"name":"Mehdi Gheisari","is_ca":true},{"name":"Mustafa Ghaderzadeh","is_ca":true},{"name":"Amir Shahab Shahabi","is_ca":false},{"name":"Zongda Wu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04931219236959951,"gpt":0.3094819476933262,"spread":0.2601697553237267,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006206296,0.0008233437,0.0007244822,0.001811987,0.0006673686,0.001094075,0.001024357,0.00129214,0.005158818],"category_scores_gemma":[0.002448098,0.0002291545,0.0006030879,0.001813399,0.0003055074,0.002656406,0.0007310713,0.001283318,0.003326725],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009054172,"about_ca_system_score_gemma":0.001572894,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007583996,"about_ca_topic_score_gemma":0.00974217,"domain_scores_codex":[0.9995552,0.00007035082,0.00004527409,0.0001395463,0.0001268222,0.0000628143],"domain_scores_gemma":[0.9990849,0.0003254322,0.00007557225,0.0001244204,0.0003266997,0.00006282092],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002427495,0.0001341855,0.0007072202,0.0001461462,0.00004037334,0.00009943236,0.00004500004,0.02009593,0.02194989,0.004845452,0.0138783,0.9378154],"study_design_scores_gemma":[0.0000180675,0.00007866747,0.000664972,0.00002412015,0.00003873727,0.0001180951,0.00003939693,0.9720643,0.01342392,0.008044233,0.005468566,0.00001688484],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04909409,0.00225709,0.9276817,0.001263351,0.001143188,0.0001620683,0.001718087,0.009320917,0.007359538],"genre_scores_gemma":[0.4653713,0.001210194,0.5054748,0.0007186317,0.0007777089,0.0002237309,0.004140392,0.0004296021,0.02165369],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007583996,"threshold_uncertainty_score":0.01725793,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4398243240","doi":"10.1016/j.datak.2024.102324","title":"Large language models: Expectations for semantics-driven systems engineering","year":2024,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Systems Engineering Methodologies and Applications","field":"Engineering","cited_by":49,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"Foundation for Research and Technology-Hellas","keywords":"Computer science; Semantics (computer science); Programming language; Modeling language; Operational semantics; Software engineering","authors":[{"name":"Robert Andrei Buchmann","is_ca":false},{"name":"Johann Eder","is_ca":false},{"name":"Hans-Georg Fill","is_ca":false},{"name":"Ulrich Frank","is_ca":false},{"name":"Dimitris Karagiannis","is_ca":false},{"name":"Emanuele Laurenzi","is_ca":false},{"name":"John Mylopoulos","is_ca":true},{"name":"Dimitris Plexousakis","is_ca":false},{"name":"Maribel Yasmina Santos","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07354628162855453,"gpt":0.3157645917063224,"spread":0.2422183100777679,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0244144,0.001301969,0.002715581,0.001996589,0.001956082,0.009456662,0.004732929,0.005453031,0.006802343],"category_scores_gemma":[0.1558973,0.001811753,0.002550144,0.001357311,0.008140259,0.02642926,0.004648992,0.00978821,0.002339423],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003920663,"about_ca_system_score_gemma":0.004205683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004967204,"about_ca_topic_score_gemma":0.003355515,"domain_scores_codex":[0.984323,0.007637258,0.0007624787,0.001633851,0.005026264,0.0006171371],"domain_scores_gemma":[0.8151162,0.1473886,0.003599877,0.01867912,0.012117,0.003099177],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001746812,0.0001374414,0.0009831304,0.0002416883,0.00007099189,0.0000779393,0.0004978883,0.02663882,0.000860552,0.9520926,0.003748097,0.01447607],"study_design_scores_gemma":[0.00003399764,0.00002493722,0.0001003003,0.00004243642,0.00001920198,0.00003431596,0.00008745705,0.07095242,0.0005204681,0.9251095,0.003055513,0.00001948689],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01461187,0.0009548896,0.9553972,0.01592303,0.0003209646,0.0002000412,0.0005269573,0.001043607,0.01102136],"genre_scores_gemma":[0.4954734,0.001790847,0.4873873,0.004724533,0.00148847,0.001367502,0.00203047,0.001225438,0.00451211],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0244144,"threshold_uncertainty_score":0.1291173,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2011939484","doi":"10.1016/j.datak.2014.09.004","title":"Privacy-preserving trajectory stream publishing","year":2014,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University; Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Data publishing; Trajectory; Computer science; Publishing; Internet privacy; World Wide Web; Political science; Law; Physics","authors":[{"name":"Khalil Al-Hussaeni","is_ca":true},{"name":"Benjamin C. M. Fung","is_ca":true},{"name":"William K. Cheung","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04019297403125713,"gpt":0.2642967261602062,"spread":0.2241037521289491,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004296929,0.0005735715,0.001672993,0.001435704,0.001465793,0.005317653,0.002806374,0.001828798,0.003489816],"category_scores_gemma":[0.02685609,0.0006189529,0.001118726,0.004742379,0.001410516,0.007693193,0.004664967,0.003037205,0.002137824],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001044986,"about_ca_system_score_gemma":0.003149205,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001066008,"about_ca_topic_score_gemma":0.000733883,"domain_scores_codex":[0.9956046,0.001051245,0.000505386,0.0007157988,0.001680937,0.0004420889],"domain_scores_gemma":[0.9795635,0.004852523,0.0009489279,0.01250774,0.00172861,0.0003987171],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001282694,0.0002449279,0.005750312,0.0003295584,0.0002154323,0.0009322385,0.0006455805,0.134086,0.008052793,0.5402718,0.02139606,0.2867926],"study_design_scores_gemma":[0.0001067665,0.0001217932,0.0006402496,0.00005960772,0.00008975974,0.001056358,0.0002648761,0.5275335,0.02071649,0.4278586,0.02151093,0.00004109312],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02093382,0.0005567865,0.968127,0.002084312,0.0003360628,0.0001046847,0.001373781,0.001389395,0.005094146],"genre_scores_gemma":[0.8055511,0.001838681,0.1743348,0.0006621237,0.000761645,0.0001995881,0.004008459,0.000351341,0.01229232],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005317653,"threshold_uncertainty_score":0.02272463,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2081024207","doi":"10.1016/j.datak.2007.05.006","title":"A tree-projection-based algorithm for multi-label recurrent-item associative-classification rule generation","year":2007,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Association rule learning; Associative property; Computer science; Data mining; Apriori algorithm; Artificial intelligence; Classifier (UML); Pattern recognition (psychology); Machine learning; Projection (relational algebra); Algorithm; Mathematics","authors":[{"name":"Rafał Rak","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true},{"name":"Marek Reformat","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1526888925603266,"gpt":0.3559568294766817,"spread":0.2032679369163551,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002348762,0.00136437,0.002449723,0.002762439,0.001289417,0.001711335,0.003328829,0.0019954,0.009301401],"category_scores_gemma":[0.006256241,0.000819456,0.001231011,0.003650088,0.0007325072,0.001965451,0.002138587,0.002842054,0.004691981],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007422748,"about_ca_system_score_gemma":0.003099323,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007556651,"about_ca_topic_score_gemma":0.01272731,"domain_scores_codex":[0.9983784,0.0003489142,0.00017027,0.0004370304,0.0005524334,0.0001128201],"domain_scores_gemma":[0.9967759,0.001580551,0.0001047589,0.0003334512,0.001090189,0.0001150637],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000150347,0.0002081701,0.0005802533,0.0001069093,0.00009872235,0.00008923143,0.0000786567,0.01412508,0.003602045,0.003301777,0.00700027,0.9706585],"study_design_scores_gemma":[0.0001097322,0.0001245841,0.0005454697,0.00003935627,0.00008862577,0.0003106161,0.00005820052,0.9746383,0.007766826,0.01210261,0.004176777,0.00003882274],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003464374,0.0001894014,0.9913722,0.000115112,0.00006046235,0.0002303844,0.0002094276,0.003758579,0.000600033],"genre_scores_gemma":[0.02962282,0.0001024547,0.9674339,0.000127181,0.00003891646,0.0003723502,0.0006861137,0.0001393312,0.001476916],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009301401,"threshold_uncertainty_score":0.03111625,"prediction_status":"machine_predicted_unvalidated"},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"}],"label_agreement":"split"},{"id":"W2068851374","doi":"10.1016/j.datak.2010.01.001","title":"Towards an accurate functional size measurement procedure for conceptual models in an MDA environment","year":2010,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Service-Oriented Architecture and Web Services","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"","keywords":"Function point; Computer science; Measure (data warehouse); Function (biology); Software; Phase (matter); Data mining; Software development; Programming language; Physics","authors":[{"name":"Beatriz Marín","is_ca":false},{"name":"Óscar Pastor","is_ca":false},{"name":"Alain Abran","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06113214184253724,"gpt":0.2597888943664013,"spread":0.1986567525238641,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0143504,0.001219256,0.001027097,0.004813447,0.00109693,0.004694655,0.002524912,0.001626271,0.001783104],"category_scores_gemma":[0.07634542,0.001196949,0.001238074,0.002031049,0.001395288,0.005999883,0.003349843,0.002761375,0.001076231],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001740883,"about_ca_system_score_gemma":0.003249373,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004065884,"about_ca_topic_score_gemma":0.004338401,"domain_scores_codex":[0.9841803,0.005083246,0.001766107,0.00116761,0.007321455,0.0004812504],"domain_scores_gemma":[0.9434285,0.01913053,0.004202194,0.01531742,0.01729591,0.0006253636],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002808479,0.0004689683,0.01937363,0.0006082053,0.0002398703,0.0002038004,0.002369711,0.07512468,0.08878896,0.1352239,0.00545167,0.6718658],"study_design_scores_gemma":[0.00006186817,0.000396976,0.009640508,0.0002218985,0.0001794129,0.0002753407,0.0008500298,0.8176202,0.09593045,0.06274381,0.01191846,0.0001610365],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008620186,0.00002685911,0.9878357,0.00006077202,0.00001120143,0.0001041118,0.00008198532,0.002640856,0.0006183641],"genre_scores_gemma":[0.1377437,0.00006076747,0.8602613,0.00004602653,0.00001133043,0.0003563766,0.0003992648,0.0006573542,0.0004640566],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0143504,"threshold_uncertainty_score":0.07589304,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2079796778","doi":"10.1016/j.datak.2013.03.005","title":"Subject-based semantic document clustering for digital forensic investigations","year":2013,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Digital and Cyber Forensics","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University","funders":"","keywords":"Subject (documents); Computer science; Cluster analysis; Information retrieval; Document clustering; Natural language processing; Artificial intelligence; Data science; World Wide Web","authors":[{"name":"Gaby G. Dagher","is_ca":true},{"name":"Benjamin C. M. Fung","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02300508494314149,"gpt":0.2321864369908625,"spread":0.209181352047721,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008448762,0.0005807148,0.0007863636,0.008139671,0.001340218,0.00204681,0.0009276006,0.001051883,0.001612368],"category_scores_gemma":[0.002447172,0.0002290328,0.001152628,0.006337168,0.0005721767,0.001789851,0.001094313,0.0005333201,0.001627674],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007862542,"about_ca_system_score_gemma":0.002380087,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006236954,"about_ca_topic_score_gemma":0.01110362,"domain_scores_codex":[0.9990849,0.0001380412,0.0001106423,0.0002360342,0.0003279111,0.0001024308],"domain_scores_gemma":[0.9987091,0.0002219579,0.0001330377,0.0003106805,0.0005376413,0.00008769842],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009586053,0.0005538635,0.01166228,0.0005520848,0.0002088973,0.0002875903,0.0007223309,0.02365585,0.06290718,0.01656646,0.01316178,0.8687631],"study_design_scores_gemma":[0.00009333932,0.0004197227,0.02096792,0.0001926764,0.0004667806,0.001282677,0.002038677,0.8120723,0.06550688,0.05618059,0.04064301,0.0001354314],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1325772,0.002047897,0.8477777,0.0003175107,0.0002332499,0.0004362643,0.004327316,0.006385979,0.005896857],"genre_scores_gemma":[0.4289235,0.00103705,0.5560168,0.00008376958,0.0001452619,0.0002209128,0.009022624,0.000303366,0.004246597],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008139671,"threshold_uncertainty_score":0.01240128,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2090992497","doi":"10.1016/j.datak.2004.06.005","title":"eMarketplaces for enterprise and cross enterprise integration","year":2004,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Auction Theory and Applications","field":"Decision Sciences","cited_by":23,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"National Research Council Canada; Western University","funders":"","keywords":"Enterprise integration; Business; Process management; Enterprise software","authors":[{"name":"Hamada Ghenniwa","is_ca":true},{"name":"Michael N. Huhns","is_ca":false},{"name":"Weiming Shen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08516768738760731,"gpt":0.4081140141277604,"spread":0.3229463267401531,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006712499,0.000472906,0.0009978034,0.001483602,0.00241963,0.006630281,0.003105128,0.003641261,0.02426142],"category_scores_gemma":[0.01451547,0.000665714,0.001158013,0.002508531,0.002140397,0.01715361,0.006837648,0.00313858,0.002316108],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001459178,"about_ca_system_score_gemma":0.00194495,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00146805,"about_ca_topic_score_gemma":0.001801134,"domain_scores_codex":[0.9966724,0.001431409,0.0002349169,0.0005011425,0.0007444678,0.0004157104],"domain_scores_gemma":[0.9934639,0.002558612,0.0005716155,0.002044985,0.0007551119,0.0006056562],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001878409,0.0001908057,0.0006819001,0.00004930749,0.00002041524,0.00008820323,0.0001249404,0.008280039,0.0004168861,0.923918,0.002518861,0.06352283],"study_design_scores_gemma":[0.00007321477,0.00008374052,0.0002446525,0.00004893606,0.00002238359,0.000170444,0.0002490754,0.08041192,0.001108208,0.9027869,0.01477367,0.00002687204],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05378284,0.000805264,0.9002603,0.003104068,0.0002832355,0.0001885254,0.0002301621,0.001072682,0.04027284],"genre_scores_gemma":[0.681456,0.0005051402,0.2929186,0.0003319313,0.0001842287,0.00023114,0.0003836201,0.0001637941,0.02382548],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02426142,"threshold_uncertainty_score":0.08116251,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2082651521","doi":"10.1016/j.datak.2006.03.005","title":"Compact access control labeling for efficient secure XML query evaluation","year":2006,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Access Control and Trust","field":"Social Sciences","cited_by":22,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; XML Signature; Access control; XML Encryption; XML database; Efficient XML Interchange; Streaming XML; XML framework; Granularity; Data access; Database; XML validation; XML; Computer network; World Wide Web; Programming language","authors":[{"name":"Huaxin Zhang","is_ca":true},{"name":"Ning Zhang","is_ca":true},{"name":"Kenneth Salem","is_ca":true},{"name":"Donghui Zhuo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06344754137285828,"gpt":0.3806975017838266,"spread":0.3172499604109683,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005700632,0.000718169,0.001905654,0.002353925,0.00173452,0.005042122,0.001659783,0.001433991,0.004705595],"category_scores_gemma":[0.02308122,0.0008466856,0.001071892,0.002122909,0.002152204,0.009105924,0.003999448,0.002533441,0.001133159],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003423285,"about_ca_system_score_gemma":0.003867605,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004814226,"about_ca_topic_score_gemma":0.005332332,"domain_scores_codex":[0.9898118,0.002994106,0.001108347,0.001183577,0.003949195,0.0009529585],"domain_scores_gemma":[0.9732084,0.01219638,0.001521747,0.009185916,0.00340661,0.0004809637],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002555388,0.0005533223,0.004541896,0.0004129556,0.0001225329,0.0003644167,0.001389572,0.06886094,0.04773636,0.2889378,0.01432178,0.570203],"study_design_scores_gemma":[0.0001243639,0.0001796859,0.000606649,0.00004635093,0.0000742618,0.0002192751,0.0002453969,0.7484897,0.04406741,0.2002989,0.005585554,0.00006239448],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03484815,0.0001723416,0.956647,0.0004164847,0.0000463842,0.0002806105,0.0004738524,0.005571128,0.001543989],"genre_scores_gemma":[0.5743365,0.0001089876,0.4204035,0.0002224756,0.00008071128,0.0003103658,0.001077453,0.0005873766,0.002872645],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005700632,"threshold_uncertainty_score":0.03014821,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2134227416","doi":"10.1016/s0169-023x(00)00044-6","title":"Selecting and materializing horizontally partitioned warehouse views","year":2001,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Windsor","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Data warehouse; Cube (algebra); Aggregate (composite); Computer science; Table (database); Warehouse; Dimension (graph theory); Data cube; Materialized view; Dimensional modeling; Database; Constraint (computer-aided design); Online analytical processing; Data mining; Information retrieval; Mathematics; View; Geography; Database design; Combinatorics","authors":[{"name":"C. I. Ezeife","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0577024843604405,"gpt":0.2827958738468403,"spread":0.2250933894863998,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001754801,0.001278239,0.001796778,0.00345595,0.001272435,0.006027469,0.002075428,0.001246686,0.007617647],"category_scores_gemma":[0.009104935,0.0016519,0.002117656,0.005021738,0.0008252127,0.005474849,0.00370615,0.001475121,0.001513354],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009373035,"about_ca_system_score_gemma":0.001846758,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003817797,"about_ca_topic_score_gemma":0.007632033,"domain_scores_codex":[0.9979618,0.0002624806,0.0001818242,0.0003196439,0.0009266374,0.0003477612],"domain_scores_gemma":[0.9961197,0.001636741,0.0002374127,0.0008420384,0.0008679302,0.0002962083],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002218115,0.0005762768,0.01383857,0.0009454864,0.0004097114,0.001734171,0.002508887,0.0211904,0.1250344,0.01504248,0.03850026,0.7780012],"study_design_scores_gemma":[0.0006141164,0.001547661,0.01766333,0.0005439307,0.001375087,0.002125824,0.01152841,0.5587093,0.2705075,0.06037564,0.07462111,0.0003882441],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.312275,0.002089538,0.6432935,0.001190537,0.000335059,0.001189275,0.003348125,0.02242979,0.01384915],"genre_scores_gemma":[0.2688026,0.0008741787,0.7168713,0.0001962243,0.0000997905,0.0002017783,0.006587169,0.002081821,0.004285245],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007617647,"threshold_uncertainty_score":0.02548361,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4388090033","doi":"10.1016/j.datak.2023.102240","title":"Hierarchical framework for interpretable and specialized deep reinforcement learning-based predictive maintenance","year":2023,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Reliability and Maintenance Optimization","field":"Engineering","cited_by":22,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"H2020 Marie Skłodowska-Curie Actions; Canadian Institute of Steel Construction; Horizon 2020 Framework Programme; Österreichische Forschungsförderungsgesellschaft; Science Foundation Ireland; Horizon 2020","keywords":"Interpretability; Reinforcement learning; Machine learning; Computer science; Artificial intelligence; Probabilistic logic; Black box; Context (archaeology); Markov decision process; Risk analysis (engineering); Markov process","authors":[{"name":"Ammar N. Abbas","is_ca":false},{"name":"Georgios C. Chasparis","is_ca":false},{"name":"John D. Kelleher","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0151796263670819,"gpt":0.2519409794108211,"spread":0.2367613530437392,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006604235,0.0007725386,0.0009458678,0.0005279849,0.0003220009,0.0008779212,0.002520048,0.001169335,0.004899127],"category_scores_gemma":[0.002049467,0.0004596637,0.0006757112,0.0004451576,0.0005899517,0.001104024,0.001152101,0.001676633,0.000789146],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001137878,"about_ca_system_score_gemma":0.001672851,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01075487,"about_ca_topic_score_gemma":0.01719184,"domain_scores_codex":[0.9996811,0.00005381981,0.00001721659,0.0001040598,0.00008196046,0.00006183515],"domain_scores_gemma":[0.9994408,0.0002229378,0.00005810237,0.00008880058,0.0001473052,0.00004193994],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001122182,0.00009750725,0.0006341479,0.0000809222,0.00004457537,0.00008462937,0.00005986927,0.8639094,0.003155795,0.02021226,0.003343307,0.1082653],"study_design_scores_gemma":[0.000003140622,0.000008387973,0.00004199818,0.000002959861,0.000003545015,0.000004313174,0.000001722086,0.9956185,0.0002360683,0.003887188,0.0001900882,0.000001976793],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01030042,0.0003218107,0.9854949,0.000167126,0.00004230088,0.00003097314,0.0002245102,0.001593722,0.001824295],"genre_scores_gemma":[0.7552218,0.0002959949,0.2379251,0.0002112266,0.00008643461,0.0001721658,0.0007470979,0.0002407521,0.005099514],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01075487,"threshold_uncertainty_score":0.02138454,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2025155398","doi":"10.1016/j.datak.2009.10.001","title":"Skyline queries with constraints: Integrating skyline and traditional query operators","year":2009,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Calgary","funders":"","keywords":"Skyline; Computer science; Pruning; Class (philosophy); Space (punctuation); Operator (biology); Database; Computation; Information retrieval; Data mining; Theoretical computer science; Algorithm; Artificial intelligence","authors":[{"name":"Ming Zhang","is_ca":true},{"name":"Reda Alhajj","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02709674014065473,"gpt":0.2344229936000985,"spread":0.2073262534594438,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005161947,0.001047571,0.002088022,0.00206789,0.0008346182,0.003783,0.00361251,0.001102574,0.00400952],"category_scores_gemma":[0.01276149,0.0007389361,0.00106857,0.005667597,0.001130295,0.01091166,0.004595242,0.00148146,0.000847752],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007951144,"about_ca_system_score_gemma":0.001574767,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004289238,"about_ca_topic_score_gemma":0.008845294,"domain_scores_codex":[0.9949195,0.001276135,0.0005099623,0.0007555536,0.002289212,0.0002496568],"domain_scores_gemma":[0.990529,0.003890262,0.0007482592,0.002830372,0.001575641,0.0004264128],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002272796,0.0006525393,0.009341892,0.001955671,0.0005765926,0.0008272638,0.001435663,0.05770215,0.02322547,0.1114787,0.0823155,0.7082158],"study_design_scores_gemma":[0.0003778154,0.0003532865,0.001790763,0.0001665985,0.0002665593,0.0008587037,0.0006025313,0.7883137,0.01802688,0.11489,0.07421883,0.0001343086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0148405,0.001031833,0.9720415,0.0005767847,0.00007530432,0.0001646072,0.001328311,0.007505117,0.002436156],"genre_scores_gemma":[0.1648113,0.001145198,0.8260269,0.0003999732,0.0002316358,0.0001830104,0.003568288,0.001783587,0.001850135],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005161947,"threshold_uncertainty_score":0.02729928,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2036207056","doi":"10.1016/s0169-023x(01)00018-0","title":"Using information extraction and natural language generation to answer e-mail","year":2001,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language generation; Question answering; Domain (mathematical analysis); Natural language; Information extraction; Text messaging; Natural language processing; Frequently asked questions; Information retrieval; World Wide Web; Artificial intelligence","authors":[{"name":"Leila Kosseim","is_ca":true},{"name":"Stéphane Beauregard","is_ca":true},{"name":"Guy Lapalme","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03129771263752603,"gpt":0.3164762397476485,"spread":0.2851785271101225,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002780384,0.001054931,0.000976931,0.003698568,0.001032425,0.00197812,0.001435614,0.001654337,0.00582397],"category_scores_gemma":[0.01289057,0.0005775355,0.001310918,0.001964611,0.0006801005,0.003250058,0.001367671,0.001259403,0.003226553],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007025282,"about_ca_system_score_gemma":0.00145477,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001880572,"about_ca_topic_score_gemma":0.002501783,"domain_scores_codex":[0.99683,0.001473991,0.0002926204,0.0005454548,0.0007138678,0.0001441296],"domain_scores_gemma":[0.9907872,0.00649669,0.0004194024,0.0006623454,0.001527278,0.0001070182],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005463134,0.00072803,0.003989737,0.0009534042,0.0001638859,0.001056473,0.001132382,0.01221112,0.0362543,0.0286459,0.03061978,0.8836987],"study_design_scores_gemma":[0.000367475,0.0004230545,0.002594876,0.0002225447,0.0004868101,0.001352301,0.0007958126,0.6934906,0.1192533,0.12047,0.06039759,0.0001457752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03219078,0.0005265296,0.9417197,0.001223827,0.0002529606,0.0006976412,0.001444495,0.01606913,0.005875005],"genre_scores_gemma":[0.1615443,0.0003226533,0.8284907,0.0004324634,0.0001694623,0.0003953925,0.004551901,0.0004470424,0.003646058],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00582397,"threshold_uncertainty_score":0.01948315,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4386987509","doi":"10.1016/j.datak.2023.102229","title":"Robotic process automation using process mining — A systematic literature review","year":2023,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Robotic Process Automation Applications","field":"Engineering","cited_by":20,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; University of Ottawa","keywords":"Computer science; Automation; Process (computing); Preprocessor; Process mining; Event (particle physics); Software engineering; Domain (mathematical analysis); Software; Data science; Construct (python library); Robot; Data mining; Work in process; Business process; Artificial intelligence; Engineering; Business process modeling","authors":[{"name":"Najah Mary El-Gharib","is_ca":true},{"name":"Daniel Amyot","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03681758029791506,"gpt":0.3163027075077396,"spread":0.2794851272098245,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01232549,0.001777957,0.005724599,0.01553965,0.0009765257,0.003565264,0.002547091,0.00201663,0.002896077],"category_scores_gemma":[0.0433202,0.001042583,0.006300724,0.01221722,0.001210272,0.003776319,0.002305681,0.001371457,0.0004524592],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002252543,"about_ca_system_score_gemma":0.01783134,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005025879,"about_ca_topic_score_gemma":0.01549508,"domain_scores_codex":[0.9928724,0.001814273,0.003022573,0.0006874543,0.001419969,0.0001833574],"domain_scores_gemma":[0.9607456,0.02831145,0.005884446,0.0009363991,0.003736609,0.0003854375],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0002076853,0.0001078355,0.001819575,0.8268255,0.005722668,0.0002268326,0.0004948615,0.0003801933,0.000450886,0.0007914446,0.00260258,0.1603699],"study_design_scores_gemma":[0.0001931514,0.0003379876,0.004278195,0.8989308,0.04743007,0.000796601,0.000796792,0.0003888522,0.0006374351,0.001298708,0.04481803,0.0000932314],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.001699176,0.9951077,0.001250003,0.0003367214,0.0001155184,0.0005462418,0.0004639223,0.00001784136,0.0004628584],"genre_scores_gemma":[0.01500664,0.9794914,0.00360616,0.0005754369,0.00008393114,0.000650743,0.0004279472,0.00001220865,0.0001455433],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01553965,"threshold_uncertainty_score":0.06518424,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2002413229","doi":"10.1016/j.datak.2012.09.005","title":"Comparing functionality of software systems: An ontological approach","year":2012,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Software system; Software engineering; Adaptation (eye); Function (biology); USable; Software development; Software; Software metric; Software quality; Point (geometry); Software construction; Function point; Quality (philosophy); Software sizing; World Wide Web; Programming language","authors":[{"name":"Iris Reinhartz-Berger","is_ca":false},{"name":"Arnon Sturm","is_ca":false},{"name":"Yair Wand","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1396288150849168,"gpt":0.3143660971102691,"spread":0.1747372820253524,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007802832,0.0006332292,0.0006532074,0.00963484,0.00193206,0.007088822,0.002571203,0.001677986,0.00187896],"category_scores_gemma":[0.02525327,0.0005982196,0.002699402,0.005578896,0.004620921,0.01198543,0.003426857,0.001371692,0.0002946997],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002027139,"about_ca_system_score_gemma":0.003040523,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006202001,"about_ca_topic_score_gemma":0.00455643,"domain_scores_codex":[0.992191,0.00305108,0.001186023,0.0006400142,0.002470171,0.0004617952],"domain_scores_gemma":[0.9826915,0.01051263,0.001027048,0.003561221,0.001749974,0.0004577295],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001662761,0.0002323683,0.01876195,0.0006292909,0.0002661223,0.0005868811,0.007017516,0.01403598,0.01252217,0.7852103,0.001115715,0.1594556],"study_design_scores_gemma":[0.00005457006,0.0003088634,0.01794397,0.0008485077,0.0009353033,0.0009294883,0.01106101,0.09883818,0.01260446,0.8239443,0.03238694,0.0001444259],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1664649,0.0005929579,0.8076092,0.001260496,0.00007145385,0.0003142417,0.0009699265,0.0006111232,0.02210567],"genre_scores_gemma":[0.6113967,0.0005047121,0.3848816,0.000166615,0.00003227603,0.0001869555,0.001641545,0.0001322128,0.001057315],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00963484,"threshold_uncertainty_score":0.04126579,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2094060849","doi":"10.1016/j.datak.2010.03.007","title":"Ranking bias in deep web size estimation using capture recapture method","year":2010,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Windsor","funders":"Natural Sciences and Engineering Research Council of Canada; State Key Laboratory of Novel Software Technology","keywords":"Ranking (information retrieval); Computer science; Estimation; Sampling (signal processing); Data mining; Process (computing); Matching (statistics); Mark and recapture; Rank (graph theory); Limit (mathematics); Statistics; Information retrieval; Mathematics","authors":[{"name":"Jianguo Lü","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04137654040511619,"gpt":0.3090537414834847,"spread":0.2676772010783685,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01633218,0.0006481474,0.001334238,0.002630101,0.0007127601,0.001376731,0.003413481,0.001718978,0.001395158],"category_scores_gemma":[0.04658943,0.0008022934,0.001106185,0.001950856,0.001113598,0.002739721,0.001404478,0.001309363,0.0004684526],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009392276,"about_ca_system_score_gemma":0.0006469178,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004529726,"about_ca_topic_score_gemma":0.005108615,"domain_scores_codex":[0.9954774,0.002397901,0.0001695105,0.0009810714,0.0006238922,0.0003504006],"domain_scores_gemma":[0.9398468,0.04928198,0.002936032,0.005453202,0.002061409,0.0004206027],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004939754,0.0002090281,0.1513406,0.0003782659,0.001305043,0.0005498679,0.0003865747,0.4513527,0.007002406,0.06371681,0.006325785,0.3169389],"study_design_scores_gemma":[0.00001674243,0.00005036124,0.0109271,0.00001854077,0.00009993631,0.0001244517,0.00003584382,0.9651141,0.002610695,0.02025442,0.0007128011,0.00003503654],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1211228,0.0004393273,0.8766174,0.0001859157,0.00005889485,0.00003705051,0.0001817816,0.0005524368,0.000804336],"genre_scores_gemma":[0.8827783,0.000195953,0.1128196,0.0002302114,0.0001107628,0.0001023266,0.0006749718,0.0001551392,0.002932728],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01633218,"threshold_uncertainty_score":0.08637387,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2073874433","doi":"10.1016/j.datak.2008.06.007","title":"Data privacy protection in multi-party clustering","year":2008,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"McMaster University","keywords":"Cluster analysis; Privacy protection; Internet privacy; Computer security; Data Protection Act 1998; Information privacy; Computer science; Business; Artificial intelligence","authors":[{"name":"Weijia Yang","is_ca":false},{"name":"Shangteng Huang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1658117584570718,"gpt":0.315461514927751,"spread":0.1496497564706792,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01119543,0.0005343956,0.001963206,0.001465868,0.00363688,0.005339757,0.004030563,0.003733231,0.001937031],"category_scores_gemma":[0.04045887,0.001103205,0.001874931,0.003445924,0.003510257,0.009960263,0.008182185,0.004365894,0.0007699479],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002307156,"about_ca_system_score_gemma":0.002394963,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008215068,"about_ca_topic_score_gemma":0.0005722588,"domain_scores_codex":[0.9817858,0.007947638,0.001112522,0.002529364,0.005363334,0.00126118],"domain_scores_gemma":[0.9476671,0.01914139,0.002821391,0.02737561,0.002338246,0.0006561311],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001252253,0.0002685916,0.005275837,0.0003632187,0.0003269753,0.0006337591,0.001791185,0.1944567,0.01033148,0.6305198,0.005455856,0.1493243],"study_design_scores_gemma":[0.00005242116,0.0000776892,0.000769203,0.00005707192,0.00007357935,0.0007135443,0.0003670599,0.5060968,0.01517906,0.4717212,0.00484672,0.00004558296],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03608429,0.0004377217,0.9578888,0.001636445,0.00006459554,0.00008893322,0.0002285891,0.0002959636,0.0032746],"genre_scores_gemma":[0.8778353,0.0003266303,0.1174221,0.0003474134,0.0001137024,0.0001345496,0.0002943512,0.00008857095,0.003437403],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01119543,"threshold_uncertainty_score":0.05920774,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2073390485","doi":"10.1016/j.datak.2004.10.001","title":"Extracting conceptual relationships from specialized documents","year":2004,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Variety (cybernetics); Granularity; Data science; Conceptual model; Representation (politics); Management science; Scientific modelling; Conceptual framework; Knowledge management; Data mining; Artificial intelligence; Database; Engineering","authors":[{"name":"Bowen Hui","is_ca":true},{"name":"Eric Yu","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09260547872331132,"gpt":0.2944326736658593,"spread":0.201827194942548,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001162736,0.001123777,0.001292217,0.01075134,0.001401631,0.00342466,0.001548858,0.001741321,0.005136579],"category_scores_gemma":[0.009829368,0.0008987123,0.001760493,0.01109369,0.0006845728,0.006271967,0.002271423,0.001812152,0.003180283],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001128833,"about_ca_system_score_gemma":0.00308841,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004374733,"about_ca_topic_score_gemma":0.006915447,"domain_scores_codex":[0.9988114,0.0001947909,0.0002076684,0.000274869,0.0004064497,0.0001048473],"domain_scores_gemma":[0.9951526,0.002285271,0.0004139931,0.0009725249,0.0009807583,0.0001948635],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003769334,0.0006635707,0.009350288,0.002313548,0.0002780306,0.003283299,0.002726171,0.005776055,0.0463313,0.05674441,0.02655772,0.8455987],"study_design_scores_gemma":[0.0003632379,0.0006074214,0.02930201,0.00184915,0.002472528,0.008845833,0.00867554,0.2469189,0.1178095,0.25669,0.3261615,0.0003044646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2015072,0.003844595,0.7502148,0.001723849,0.0003215769,0.001138704,0.01619167,0.007551057,0.01750649],"genre_scores_gemma":[0.2354429,0.002856883,0.7090181,0.000232217,0.0001604255,0.0003147708,0.04660063,0.0006619185,0.004712262],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01075134,"threshold_uncertainty_score":0.01718354,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4411507372","doi":"10.1016/j.datak.2025.102480","title":"Large language models for conceptual modeling: Assessment and application potential","year":2025,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Memorial University of Newfoundland","funders":"Fundação para a Ciência e a Tecnologia; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science","authors":[{"name":"Veda C. Storey","is_ca":false},{"name":"Óscar Pastor","is_ca":false},{"name":"Giancarlo Guizzardi","is_ca":false},{"name":"Stephen W. Liddle","is_ca":false},{"name":"Wolfgang Maaß","is_ca":false},{"name":"Jeffrey Parsons","is_ca":true},{"name":"Jolita Ralyté","is_ca":false},{"name":"Maribel Yasmina Santos","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02950059442361724,"gpt":0.3180202898758975,"spread":0.2885196954522802,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01213449,0.0008921156,0.001133527,0.003050379,0.001217599,0.005642711,0.002953382,0.001848339,0.005561384],"category_scores_gemma":[0.06222919,0.0008553507,0.001807612,0.003184738,0.001463775,0.01281432,0.002199812,0.002262287,0.001059254],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003368768,"about_ca_system_score_gemma":0.003464676,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01013379,"about_ca_topic_score_gemma":0.01280363,"domain_scores_codex":[0.9933918,0.004382949,0.0002680265,0.0004454928,0.001399031,0.0001128512],"domain_scores_gemma":[0.9219537,0.06500807,0.001323809,0.006596714,0.004282078,0.0008356578],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009196377,0.0008963728,0.009566668,0.0009416882,0.0007076706,0.0002477349,0.001305461,0.2486024,0.002578216,0.502632,0.01408407,0.2175181],"study_design_scores_gemma":[0.00007220844,0.00008645376,0.0006325508,0.0001315609,0.0001482681,0.00008062735,0.0002610312,0.7609238,0.001050743,0.2284565,0.008116593,0.00003972879],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05721329,0.002398024,0.925136,0.003983676,0.0001655694,0.0003011371,0.00191638,0.002755243,0.006130578],"genre_scores_gemma":[0.4411584,0.002254696,0.5487381,0.0004827335,0.0002037658,0.0007587356,0.003551555,0.0007204637,0.002131542],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01213449,"threshold_uncertainty_score":0.06417412,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3197889602","doi":"10.1016/j.datak.2021.101924","title":"Mining high utility patterns in interval-based event sequences","year":2021,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Regina","funders":"","keywords":"Pruning; Interval (graph theory); Computer science; Event (particle physics); Data mining; Range (aeronautics); Point (geometry); Sequential Pattern Mining; Algorithm; Mathematics; Engineering","authors":[{"name":"S. Mohammad Mirbagheri","is_ca":true},{"name":"Howard J. Hamilton","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04147215521048864,"gpt":0.2906286266056325,"spread":0.2491564713951439,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009808433,0.0005776577,0.0005798145,0.003808027,0.0004176025,0.00120579,0.0008760173,0.0005731884,0.001499646],"category_scores_gemma":[0.008804132,0.0002744858,0.0005805909,0.003324162,0.0003671436,0.001641637,0.0006820198,0.0007860485,0.0004677708],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003130095,"about_ca_system_score_gemma":0.0005242164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001173197,"about_ca_topic_score_gemma":0.001693078,"domain_scores_codex":[0.9986712,0.000177996,0.0001664819,0.0002762836,0.0005861606,0.000121903],"domain_scores_gemma":[0.9943438,0.00329743,0.0009635175,0.0004074449,0.0007313613,0.0002565007],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0027874,0.001091513,0.2186295,0.001343855,0.000547819,0.005843346,0.001157056,0.101101,0.03664014,0.02924733,0.005608093,0.5960029],"study_design_scores_gemma":[0.0001372634,0.0008414179,0.05568913,0.0002608765,0.0003455096,0.004009582,0.000871182,0.822716,0.02271456,0.08441624,0.007916087,0.00008215757],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5598643,0.001413087,0.4304648,0.0004170079,0.0001040782,0.0002064264,0.004034076,0.001116829,0.002379308],"genre_scores_gemma":[0.9074109,0.0004530348,0.0882272,0.0000499468,0.00007891272,0.00007792244,0.002799925,0.00005015945,0.0008520084],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.003808027,"threshold_uncertainty_score":0.005187213,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2114928940","doi":"10.1016/j.datak.2013.02.001","title":"Assessing the quality factors found in in-line documentation written in natural language: The JavadocMiner","year":2013,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University","funders":"U.S. Department of Defense","keywords":"Documentation; Natural (archaeology); Quality (philosophy); Line (geometry); Natural language processing; Computer science; Linguistics; Psychology; History; Mathematics; Programming language; Archaeology; Philosophy","authors":[{"name":"Ninus Khamis","is_ca":true},{"name":"Juergen Rilling","is_ca":true},{"name":"René Witte","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05909199711209612,"gpt":0.3912896514703549,"spread":0.3321976543582588,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008280525,0.0004564941,0.0005453289,0.003678482,0.0004767217,0.00213692,0.0008031995,0.0008680683,0.0009938509],"category_scores_gemma":[0.06522698,0.0003589376,0.0003747304,0.002650032,0.0003661556,0.001648852,0.001134989,0.0006043764,0.0005257657],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005909993,"about_ca_system_score_gemma":0.001580102,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002810807,"about_ca_topic_score_gemma":0.004964848,"domain_scores_codex":[0.9932473,0.001738016,0.001245113,0.0008639572,0.002699762,0.0002058676],"domain_scores_gemma":[0.8842165,0.0609452,0.01862932,0.01018862,0.02460107,0.00141926],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002356272,0.001920891,0.3902852,0.002567362,0.0004928734,0.0008241331,0.006517689,0.006379845,0.07320432,0.001041574,0.008980397,0.5054296],"study_design_scores_gemma":[0.0006299986,0.00380947,0.6573953,0.0008421296,0.001079494,0.002640302,0.00520625,0.1296501,0.1717581,0.001963374,0.02464075,0.0003847088],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.982269,0.000276978,0.01205319,0.0001486838,0.0000163259,0.00006660742,0.001135002,0.002818464,0.00121563],"genre_scores_gemma":[0.9168324,0.0002200703,0.07473063,0.00007109977,0.00001466044,0.00007035029,0.005357023,0.00100287,0.001700897],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008280525,"threshold_uncertainty_score":0.04379213,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2052266824","doi":"10.1016/j.datak.2003.10.007","title":"Conceptual construction on incomplete survey data","year":2003,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Rough Sets and Fuzzy Logic","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Data mining; Knowledge extraction; Raw data; Data science; Conceptual model; Set (abstract data type); Information retrieval; Database","authors":[{"name":"Shouhong Wang","is_ca":false},{"name":"Hai Wang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1559070856255667,"gpt":0.294042282112188,"spread":0.1381351964866213,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01165211,0.0005742958,0.001107054,0.007155336,0.00155742,0.005359141,0.002553641,0.001468722,0.004271565],"category_scores_gemma":[0.06188249,0.0009192445,0.002216234,0.008474491,0.003156313,0.01092326,0.00457343,0.002132499,0.0005310508],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00206625,"about_ca_system_score_gemma":0.002249178,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002811671,"about_ca_topic_score_gemma":0.003212887,"domain_scores_codex":[0.9897285,0.006057027,0.0008085915,0.001255792,0.001780196,0.0003699475],"domain_scores_gemma":[0.9586286,0.02542856,0.003037895,0.008625678,0.003627667,0.0006515453],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00004843199,0.00003874363,0.003544171,0.0003437542,0.00009462567,0.0001980839,0.001819573,0.01411671,0.0004341869,0.9074542,0.002420652,0.06948688],"study_design_scores_gemma":[0.00001615596,0.00003251742,0.001993878,0.0002138945,0.00007101889,0.0002463879,0.001416536,0.07735794,0.0007984606,0.9038005,0.01401631,0.00003633176],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03009731,0.0004817229,0.9634358,0.001179319,0.00003784703,0.00008933756,0.0007933762,0.0001963071,0.003688872],"genre_scores_gemma":[0.4748297,0.00109709,0.5183493,0.0002480794,0.00007311373,0.0004457845,0.003312171,0.00009957169,0.00154519],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01165211,"threshold_uncertainty_score":0.06162298,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1983768816","doi":"10.1016/s0169-023x(02)00133-7","title":"Optimizing temporal queries: efficient handling of duplicates","year":2002,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; SQL; Query language; Query optimization; Query by Example; Relational database; Temporal database; Semantics (computer science); Expressive power; Context (archaeology); Sargable; Information retrieval; Programming language; Database; Web search query; Search engine","authors":[{"name":"Ivan T. Bowman","is_ca":true},{"name":"David Toman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03893282992648182,"gpt":0.2496358099264162,"spread":0.2107029799999343,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005367592,0.00125525,0.002358575,0.001861517,0.001633156,0.003992253,0.00483478,0.001736515,0.003166741],"category_scores_gemma":[0.02073514,0.00119091,0.001203222,0.004802144,0.00131475,0.008217559,0.003467334,0.001524926,0.001405733],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001257911,"about_ca_system_score_gemma":0.003030796,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005861953,"about_ca_topic_score_gemma":0.006236839,"domain_scores_codex":[0.991804,0.001788187,0.0009356915,0.001080777,0.003694429,0.0006969456],"domain_scores_gemma":[0.9838435,0.007553881,0.001046685,0.00454368,0.002654865,0.0003574191],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003944359,0.000632574,0.006447101,0.001027676,0.0004068358,0.0009719688,0.001336788,0.07599457,0.04906671,0.04932555,0.06717879,0.7436671],"study_design_scores_gemma":[0.0006721465,0.000436923,0.001453861,0.00008640213,0.0004363192,0.001319298,0.0007997855,0.8034211,0.06544702,0.09697572,0.02879767,0.0001538307],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1233564,0.003768716,0.8495,0.001305854,0.000471572,0.0002952661,0.001671631,0.01235684,0.007273614],"genre_scores_gemma":[0.4238589,0.001390549,0.5616745,0.0005689126,0.0004344936,0.0002147137,0.003538725,0.00203164,0.006287615],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005861953,"threshold_uncertainty_score":0.02838689,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3197885731","doi":"10.1016/j.datak.2021.101926","title":"Exploring the research landscape of data warehousing and mining based on DaWaK Conference full-text articles","year":2021,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Big Data and Business Intelligence","field":"Business, Management and Accounting","cited_by":5,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Ministry of Science and ICT, South Korea; National Research Foundation of Korea; Conselho Nacional de Desenvolvimento Científico e Tecnológico; National Natural Science Foundation of China; National Science Foundation","keywords":"Data science; Data warehouse; Computer science; Information retrieval; Data mining","authors":[{"name":"Tatsawan Timakum","is_ca":false},{"name":"Soobin Lee","is_ca":false},{"name":"Min Song","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.5264044121500385,"gpt":0.3663522096925166,"spread":0.1600522024575219,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.01762879,0.0006116536,0.001005015,0.01438117,0.002841856,0.02563578,0.001873245,0.001337802,0.009752742],"category_scores_gemma":[0.01920678,0.0007649448,0.0007316264,0.01976668,0.004490066,0.0234826,0.004751956,0.00321759,0.002519513],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004386621,"about_ca_system_score_gemma":0.008791558,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004213723,"about_ca_topic_score_gemma":0.006668338,"domain_scores_codex":[0.9939157,0.002418798,0.000497245,0.0007255723,0.001992699,0.0004499837],"domain_scores_gemma":[0.9667066,0.01976,0.00118448,0.002678969,0.007693097,0.001976868],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002293776,0.0002268415,0.004883004,0.002399714,0.0000717099,0.0002426371,0.001886186,0.002257253,0.003651064,0.4844913,0.02835117,0.4713098],"study_design_scores_gemma":[0.00003974149,0.0001610829,0.005182808,0.003045694,0.00007520243,0.0004528134,0.009362879,0.01697727,0.006605645,0.3334544,0.6244963,0.0001461961],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.07899578,0.2097084,0.2383525,0.1486906,0.00523655,0.0004868084,0.00200329,0.002664208,0.3138619],"genre_scores_gemma":[0.4027365,0.1921074,0.3458502,0.007938489,0.002739828,0.0003481352,0.002496125,0.001130837,0.04465247],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9856188,"threshold_uncertainty_score":0.09323108,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4411540031","doi":"10.1016/j.datak.2025.102482","title":"Domain knowledge in artificial intelligence: Using conceptual modeling to increase machine learning accuracy and explainability","year":2025,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Artificial intelligence; Computer science; Domain (mathematical analysis); Machine learning; Domain knowledge; Artificial Intelligence System; Mathematics","authors":[{"name":"Veda C. Storey","is_ca":false},{"name":"Jeffrey Parsons","is_ca":true},{"name":"Arturo Castellanos Bueso","is_ca":false},{"name":"Monica Chiarini Tremblay","is_ca":false},{"name":"Roman Lukyanenko","is_ca":false},{"name":"Alfred Castillo","is_ca":false},{"name":"Wolfgang Maaß","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06691927417424057,"gpt":0.3392689324753541,"spread":0.2723496583011135,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007338671,0.0007347451,0.0006889307,0.002505155,0.0006811179,0.003533021,0.001648726,0.001450935,0.001906175],"category_scores_gemma":[0.06069481,0.0004994324,0.001176546,0.002163863,0.001561232,0.01076403,0.002748304,0.002611614,0.00032529],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001403838,"about_ca_system_score_gemma":0.001498784,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00310334,"about_ca_topic_score_gemma":0.002566829,"domain_scores_codex":[0.9951349,0.002948869,0.0002924467,0.0006531387,0.0008702002,0.0001005391],"domain_scores_gemma":[0.9426749,0.04404849,0.002380742,0.007825116,0.002715843,0.0003548739],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005361215,0.0007695143,0.02553688,0.000716174,0.0005934539,0.0002034379,0.001877895,0.1872663,0.004803581,0.2758704,0.004344418,0.4974819],"study_design_scores_gemma":[0.00004374451,0.00007296571,0.00213981,0.000126749,0.0001841509,0.00006311284,0.0001597357,0.7046068,0.002884614,0.2867188,0.00296202,0.0000374906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08609538,0.00154842,0.9026132,0.003088485,0.0001374888,0.0001496391,0.0003135532,0.0007897972,0.005263992],"genre_scores_gemma":[0.7342622,0.0008528435,0.2628928,0.0003123938,0.000114189,0.0002285725,0.0004441584,0.0001433318,0.0007496318],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007338671,"threshold_uncertainty_score":0.03881109,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2017416607","doi":"10.1016/j.datak.2009.01.003","title":"Formalizing visibility characteristics in hierarchical systems","year":2009,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Service-Oriented Architecture and Web Services","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Visibility; Hierarchy; Computer science; Theoretical computer science; Coherence (philosophical gambling strategy); Artificial intelligence; Mathematics; Physics; Statistics; Optics","authors":[{"name":"Debmalya Biswas","is_ca":false},{"name":"K. Vidyasankar","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01914562073366399,"gpt":0.2609460380770489,"spread":0.2418004173433849,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005440859,0.0006482045,0.0007125832,0.002097082,0.00146953,0.005226235,0.002030536,0.002072178,0.003227331],"category_scores_gemma":[0.02347123,0.001508566,0.001808849,0.002035236,0.005808865,0.01432829,0.004765409,0.004321318,0.0003345872],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002241911,"about_ca_system_score_gemma":0.002198125,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00523557,"about_ca_topic_score_gemma":0.005407174,"domain_scores_codex":[0.9952587,0.001407874,0.0005742002,0.0006747188,0.001348905,0.0007356118],"domain_scores_gemma":[0.9761344,0.01554035,0.002015589,0.003296575,0.002234226,0.0007788921],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002727973,0.0000260484,0.001012148,0.00007010857,0.00001558727,0.000123836,0.0007717774,0.00875184,0.001308215,0.9799472,0.0003564031,0.007589549],"study_design_scores_gemma":[0.00003305785,0.00002771787,0.000491085,0.00006193609,0.00005345369,0.0001229218,0.0003769299,0.07781404,0.002434619,0.913835,0.004721788,0.00002743547],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06269003,0.000245188,0.9272367,0.0007433008,0.00005148043,0.0001001697,0.0001188825,0.0005248936,0.008289206],"genre_scores_gemma":[0.7933488,0.0002732544,0.2030458,0.0001383288,0.0001166436,0.0001477076,0.000293869,0.0002643353,0.002371231],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005440859,"threshold_uncertainty_score":0.02877438,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2079690139","doi":"10.1016/j.datak.2006.11.002","title":"Load balancing and data placement for multi-tiered database systems","year":2006,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"IBM (Canada)","funders":"","keywords":"Computer science; Database; Materialized view; View; Database server; Workload; Online analytical processing; Database tuning; Server; Table (database); IBM; Database administrator; Distributed database; Database testing; Operating system; Application server; Load balancing (electrical power); Data warehouse; Database design","authors":[{"name":"Wen‐Syan Li","is_ca":false},{"name":"Daniel C. Zilio","is_ca":true},{"name":"Vishal Batra","is_ca":false},{"name":"Calisto Zuzarte","is_ca":true},{"name":"Inderpal Narang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05706698903393938,"gpt":0.3019583151155923,"spread":0.2448913260816529,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002174549,0.0008293857,0.001133938,0.001254519,0.002088974,0.003351429,0.003729957,0.001089227,0.005064533],"category_scores_gemma":[0.007683186,0.0009684344,0.0004447995,0.003204073,0.0007915994,0.004922389,0.002464309,0.0009357112,0.000883897],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001576677,"about_ca_system_score_gemma":0.002043777,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007042919,"about_ca_topic_score_gemma":0.01407848,"domain_scores_codex":[0.9973677,0.0006508225,0.0003359278,0.0004012803,0.0007551654,0.0004891048],"domain_scores_gemma":[0.9951032,0.001330788,0.0003896233,0.001650057,0.001040132,0.0004862362],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003395468,0.0008146017,0.011481,0.0005084286,0.000226724,0.0004598467,0.0008906153,0.364609,0.05243013,0.02794983,0.03248292,0.5047514],"study_design_scores_gemma":[0.0001039061,0.0001761775,0.002117896,0.000014929,0.00006948126,0.0002232447,0.0003038031,0.962184,0.01032328,0.02082719,0.003613489,0.00004258254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2247139,0.003120413,0.7559654,0.001591401,0.0005190193,0.0002575413,0.0007640208,0.006012103,0.00705633],"genre_scores_gemma":[0.8202351,0.0003839562,0.1734515,0.0001890294,0.0001766924,0.00009128208,0.0005671654,0.0003110112,0.004594261],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007042919,"threshold_uncertainty_score":0.01694262,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4405632789","doi":"10.1016/j.datak.2024.102393","title":"Coupling MDL and Markov chain Monte Carlo to sample diverse pattern sets","year":2024,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Image Retrieval and Classification Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Markov chain Monte Carlo; Coupling (piping); Statistical physics; Monte Carlo method; Markov chain; Computer science; Sample (material); Algorithm; Physics; Mathematics; Statistics; Machine learning; Materials science; Thermodynamics","authors":[{"name":"François Camelin","is_ca":true},{"name":"Samir Loudni","is_ca":false},{"name":"Gilles Pesant","is_ca":true},{"name":"Charlotte Truchet","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03342681986294629,"gpt":0.2847333065404637,"spread":0.2513064866775174,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005190631,0.0006738356,0.002102611,0.001881586,0.0007958274,0.002080668,0.002669292,0.002238531,0.003844512],"category_scores_gemma":[0.02601424,0.001409038,0.001153947,0.0017835,0.001667201,0.002593642,0.002157411,0.00262989,0.0007915583],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00223071,"about_ca_system_score_gemma":0.002546731,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01303328,"about_ca_topic_score_gemma":0.01462481,"domain_scores_codex":[0.9983468,0.0007497497,0.0001002002,0.0002410619,0.0004242868,0.0001379089],"domain_scores_gemma":[0.973656,0.02214199,0.0008402708,0.00158553,0.001298806,0.0004774425],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005097779,0.00006587956,0.0007122814,0.0000428097,0.00004493487,0.00002534475,0.00002618874,0.9707906,0.000236037,0.01250247,0.0003541667,0.01514836],"study_design_scores_gemma":[0.000003826982,0.000002857818,0.0000192402,0.000001820209,0.000001817094,0.000002580197,0.000001258577,0.995326,0.00005357381,0.004530204,0.00005477308,0.000002142298],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01360406,0.0001471116,0.9843806,0.000203855,0.00003896388,0.00007034868,0.00007065717,0.0006515755,0.0008327788],"genre_scores_gemma":[0.5735095,0.0002114177,0.4219998,0.0004172922,0.0001284925,0.0005161089,0.0005823686,0.000397968,0.002237194],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01303328,"threshold_uncertainty_score":0.02745098,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4408912911","doi":"10.1016/j.datak.2025.102440","title":"Customized long short-term memory architecture for multi-document summarization with improved text feature set","year":2025,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Automatic summarization; Computer science; Term (time); Feature (linguistics); Set (abstract data type); Information retrieval; Architecture; Natural language processing; Artificial intelligence; Programming language; History; Linguistics","authors":[{"name":"Satya Deo","is_ca":false},{"name":"Debajyoty Banik","is_ca":true},{"name":"Prasant Kumar Pattnaik","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02347369987771302,"gpt":0.2815724607136532,"spread":0.2580987608359402,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000413646,0.0009054534,0.0008562098,0.001789634,0.0005896788,0.001072963,0.001666816,0.0006282655,0.008775624],"category_scores_gemma":[0.001019622,0.0003057845,0.0005970314,0.002269811,0.0001515924,0.001493644,0.0007630839,0.0006535944,0.004326717],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005435116,"about_ca_system_score_gemma":0.001032776,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005671014,"about_ca_topic_score_gemma":0.009160771,"domain_scores_codex":[0.9997132,0.00002908471,0.00004603561,0.00009410531,0.00007271098,0.0000448117],"domain_scores_gemma":[0.9993275,0.0001433837,0.00005049779,0.0001344003,0.0002981063,0.00004615309],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00112956,0.0003265448,0.001069774,0.0003037049,0.0001864673,0.0001763182,0.0001554628,0.008972655,0.08531071,0.001128614,0.02521253,0.8760276],"study_design_scores_gemma":[0.0002848173,0.001308431,0.00471102,0.00006506864,0.0005268432,0.0004175415,0.0002599387,0.7652031,0.187727,0.004684469,0.03466464,0.0001471538],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07622569,0.003327904,0.8527089,0.0003597753,0.0006949157,0.0003199404,0.003936369,0.05889832,0.003528323],"genre_scores_gemma":[0.3801282,0.0008823655,0.5864038,0.000524704,0.0004370463,0.0007445498,0.01335086,0.0008394371,0.01668908],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008775624,"threshold_uncertainty_score":0.02935737,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3037550867","doi":"10.1016/j.datak.2013.01.001","title":"Diamond dicing","year":2013,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université TÉLUQ; University of New Brunswick","funders":"","keywords":"Wafer dicing; Materials science; Nanotechnology; Wafer","authors":[{"name":"Hazel Webb","is_ca":true},{"name":"Daniel Lemire","is_ca":true},{"name":"Owen Kaser","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0305814111118762,"gpt":0.2466045792735479,"spread":0.2160231681616717,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0006719115,0.0008491109,0.0008076594,0.002998227,0.002039825,0.002126048,0.001552524,0.001374425,0.2076494],"category_scores_gemma":[0.002549401,0.0005317304,0.0007107699,0.001831517,0.0008522142,0.001746976,0.002514314,0.002190732,0.05262961],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007051152,"about_ca_system_score_gemma":0.001334974,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001545897,"about_ca_topic_score_gemma":0.002550814,"domain_scores_codex":[0.9988838,0.00006633622,0.00005400756,0.000219624,0.0006701996,0.000105986],"domain_scores_gemma":[0.9986926,0.000173157,0.00004374784,0.0004216087,0.0005508488,0.000118083],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002788423,0.00006525256,0.0005593763,0.0004376066,0.00002572062,0.0005022361,0.0002091069,0.002620682,0.01556131,0.1736902,0.3190831,0.4869666],"study_design_scores_gemma":[0.00004251735,0.00005846533,0.0003032098,0.00007672168,0.00001332822,0.0006701846,0.0001259633,0.004529274,0.0111586,0.02757488,0.9554117,0.00003518411],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.009290629,0.001502258,0.1970457,0.002601035,0.005889241,0.0004514878,0.002563318,0.007415484,0.7732409],"genre_scores_gemma":[0.1229949,0.001716791,0.2923267,0.002030158,0.0008671508,0.0004056504,0.005145384,0.003731426,0.5707819],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.7923506,"threshold_uncertainty_score":0.6946564,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3131740359","doi":"10.1016/j.datak.2021.101876","title":"Generation of Gaussian sets for clustering methods assessment","year":2021,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Advanced Clustering Algorithms Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Thales (Canada)","funders":"","keywords":"Cluster analysis; Computer science; Generator (circuit theory); Fuzzy clustering; Curse of dimensionality; Rand index; Mixture model; Data mining; Homogeneity (statistics); Artificial intelligence; Gaussian; Pattern recognition (psychology); Set (abstract data type); Hierarchical clustering; Maximization; Sensitivity (control systems); Algorithm; Machine learning; Mathematics; Mathematical optimization; Power (physics); Engineering","authors":[{"name":"Radhwane Gherbaoui","is_ca":false},{"name":"Mohammed Ouali","is_ca":true},{"name":"Nacéra Benamrane","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1936835263463012,"gpt":0.4732092990702447,"spread":0.2795257727239435,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009924402,0.00151307,0.001283073,0.005248564,0.001954774,0.002916367,0.002567967,0.002144465,0.005694978],"category_scores_gemma":[0.04043841,0.0007541541,0.002485624,0.002935671,0.00104456,0.001933597,0.003944206,0.002505636,0.002006472],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002120038,"about_ca_system_score_gemma":0.003760818,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005646474,"about_ca_topic_score_gemma":0.005863426,"domain_scores_codex":[0.9928108,0.002700549,0.0004044385,0.0008757592,0.002769313,0.0004391466],"domain_scores_gemma":[0.9839677,0.005843,0.0006213317,0.002344047,0.006874896,0.0003490931],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000797116,0.0004784135,0.008731701,0.0006865494,0.0003745486,0.0002634869,0.001395655,0.2284048,0.01591539,0.1112095,0.01886415,0.6128789],"study_design_scores_gemma":[0.00007129453,0.0001827875,0.002263007,0.0001266152,0.00007467827,0.0001398023,0.0002228027,0.9240623,0.01514607,0.04927241,0.008371458,0.0000666709],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0166258,0.0001336606,0.9788477,0.0001430592,0.00008049722,0.0004186926,0.0003764399,0.001584867,0.001789317],"genre_scores_gemma":[0.1725505,0.0001481566,0.8222083,0.0001000505,0.00003574939,0.0007899234,0.001777781,0.000709928,0.001679479],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009924402,"threshold_uncertainty_score":0.05248588,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4416703099","doi":"10.1016/j.datak.2025.102536","title":"Enhancing clustering stability, compactness, and separation in multimodal data environments","year":2025,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Customer churn and segmentation","field":"Business, Management and Accounting","cited_by":1,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Ministério da Ciência, Tecnologia, Inovações e Comunicações","keywords":"Cluster analysis; Benchmark (surveying); Stability (learning theory); Rand index; Hierarchical clustering; Benchmarking; Consensus clustering; Embedding; Focus (optics); Core (optical fiber)","authors":[{"name":"Fillipe dos Santos Silva","is_ca":true},{"name":"Júlio Cesar dos Reis","is_ca":true},{"name":"Marcelo S. Reis","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04420924631857913,"gpt":0.2996458049990858,"spread":0.2554365586805067,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003780584,0.001130859,0.00132678,0.002604727,0.001318053,0.002517643,0.001401821,0.001366728,0.0009416192],"category_scores_gemma":[0.0153403,0.0006539511,0.0007854956,0.002376004,0.001067937,0.003566838,0.003726916,0.001101693,0.0004215062],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001262272,"about_ca_system_score_gemma":0.001227678,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005427242,"about_ca_topic_score_gemma":0.007086963,"domain_scores_codex":[0.9977767,0.0007557273,0.0001495784,0.0004859733,0.0005691308,0.000262857],"domain_scores_gemma":[0.9913042,0.005426342,0.0006867053,0.0007415349,0.001481667,0.0003596818],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001974484,0.0007207565,0.02238409,0.0002718951,0.0003938234,0.0002581385,0.00199684,0.4792128,0.03693017,0.01038103,0.002917194,0.4425588],"study_design_scores_gemma":[0.00001748742,0.0001090199,0.003296635,0.00001126196,0.00004422143,0.0000695393,0.0003479892,0.9817245,0.008051221,0.005874475,0.000432011,0.00002159921],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.30202,0.0004089859,0.6946422,0.0003330341,0.00003014343,0.00006477889,0.0001400873,0.0009149233,0.001445743],"genre_scores_gemma":[0.8682044,0.0001547877,0.1294361,0.00007271503,0.00005533324,0.00006062591,0.0004950973,0.0002240864,0.00129688],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005427242,"threshold_uncertainty_score":0.0199939,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4387842356","doi":"10.1016/j.datak.2023.102238","title":"User-generated short-text classification using cograph editing-based network clustering with an application in invoice categorization","year":2023,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Complex Network Analysis Techniques","field":"Physics and Astronomy","cited_by":1,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"McMaster University","funders":"Ontario Ministry of Research and Innovation; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Mitacs; Canada Foundation for Innovation","keywords":"Computer science; Categorization; Cluster analysis; Invoice; Information retrieval; Product (mathematics); Natural language processing; Data mining; Artificial intelligence; World Wide Web","authors":[{"name":"Dewan F. Wahid","is_ca":true},{"name":"Elkafi Hassini","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04509888621521016,"gpt":0.2994695044326596,"spread":0.2543706182174494,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009462829,0.001013583,0.0006067774,0.005946299,0.001031578,0.001231321,0.001099317,0.0009140779,0.008767792],"category_scores_gemma":[0.005891817,0.0002604436,0.000658419,0.004037126,0.0002929949,0.001516939,0.001015119,0.0006202015,0.004681647],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007205259,"about_ca_system_score_gemma":0.0009890049,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006521841,"about_ca_topic_score_gemma":0.01584067,"domain_scores_codex":[0.9991249,0.0001485418,0.00007511684,0.0003069835,0.0002798755,0.00006449297],"domain_scores_gemma":[0.9965464,0.00155489,0.0001787758,0.0005191553,0.001039292,0.0001615119],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00130249,0.0006644445,0.01213049,0.001156724,0.0001813145,0.0009288308,0.001408558,0.02093931,0.03688588,0.00630237,0.1131095,0.8049901],"study_design_scores_gemma":[0.000111699,0.0001761282,0.01250878,0.00009296265,0.0001058004,0.0004911462,0.0008016844,0.856596,0.04326424,0.01347557,0.07224385,0.0001320791],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2380017,0.0008916503,0.50752,0.000854451,0.001035988,0.001070785,0.0540631,0.1814596,0.0151027],"genre_scores_gemma":[0.3169282,0.0002639541,0.6100978,0.0001451334,0.0001625053,0.0005513955,0.05507698,0.003419159,0.01335493],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008767792,"threshold_uncertainty_score":0.02933115,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4413052970","doi":"10.1016/j.datak.2025.102500","title":"Corrigendum to “Domain knowledge in artificial intelligence: Using conceptual modeling to increase machine learning accuracy and explainability” [Knowledge and Data Engineering Volume 160, November 2025, 102482]","year":2025,"lang":"en","type":"erratum","venue":"Data & Knowledge Engineering","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Computer science; Domain knowledge; Artificial intelligence; Domain (mathematical analysis); Volume (thermodynamics); Machine learning; Knowledge engineering; Mathematics","authors":[{"name":"Veda C. Storey","is_ca":false},{"name":"Jeffrey Parsons","is_ca":true},{"name":"Arturo Castellanos Bueso","is_ca":false},{"name":"Monica Chiarini Tremblay","is_ca":false},{"name":"Roman Lukyanenko","is_ca":false},{"name":"Alfred Castillo","is_ca":false},{"name":"Wolfgang Maaß","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08647812660475818,"gpt":0.3363630339300379,"spread":0.2498849073252797,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004259335,0.002642672,0.0030433,0.00586689,0.007568601,0.008489485,0.004491275,0.01237486,0.1408286],"category_scores_gemma":[0.0399105,0.001332977,0.003144266,0.003876287,0.00214527,0.003219633,0.003237111,0.01018307,0.110792],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01052963,"about_ca_system_score_gemma":0.008023717,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.09917548,"about_ca_topic_score_gemma":0.1613248,"domain_scores_codex":[0.9920623,0.0009714846,0.0009115919,0.00106148,0.004258444,0.0007347799],"domain_scores_gemma":[0.9704066,0.003961705,0.0006850049,0.001434709,0.02261305,0.0008990181],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000009688247,0.000004246175,0.00001506626,0.0000221536,0.000003156561,0.00002087616,0.00000443493,0.00001690521,0.00001383277,0.0003853924,0.9979533,0.001550923],"study_design_scores_gemma":[0.00003598976,0.00002441195,0.001016569,0.0001926975,0.00003783684,0.00005920267,0.00006341848,0.0004895673,0.0002274083,0.002662021,0.995147,0.00004383875],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"other","genre_scores_codex":[0.0001315579,0.001344219,0.0008203429,0.08883569,0.8841364,0.00008309921,0.001937341,0.0006843386,0.02202701],"genre_scores_gemma":[0.004733262,0.00362777,0.002866526,0.1578751,0.1541755,0.0003142409,0.004436363,0.00118255,0.6707886],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.1408286,"threshold_uncertainty_score":0.4711185,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4390245620","doi":"10.1016/j.datak.2023.102274","title":"Mining Keys for Graphs","year":2023,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"McMaster University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Graph; Data deduplication; Theoretical computer science; Identification (biology); Data mining; Pruning; Database","authors":[{"name":"Morteza Alipourlangouri","is_ca":true},{"name":"Fei Chiang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3915863678661295,"gpt":0.4561268436149644,"spread":0.06454047574883487,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001808456,0.001388595,0.001067236,0.008389687,0.001729383,0.004390685,0.001869649,0.001415432,0.009535979],"category_scores_gemma":[0.0251241,0.001056395,0.002478394,0.00659231,0.00138712,0.01894497,0.003869871,0.002289438,0.004343549],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001304374,"about_ca_system_score_gemma":0.002173668,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002070138,"about_ca_topic_score_gemma":0.003540249,"domain_scores_codex":[0.9965525,0.0004445231,0.0004067327,0.001302763,0.0009276579,0.0003657693],"domain_scores_gemma":[0.9858899,0.007058039,0.001245165,0.003475915,0.001820456,0.0005104181],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0007477858,0.0003087108,0.01917157,0.002373173,0.0004108478,0.00108398,0.001795429,0.0165182,0.01670516,0.3431718,0.07495227,0.5227611],"study_design_scores_gemma":[0.00007423409,0.0001504275,0.003083076,0.0004407937,0.0003173493,0.001265789,0.0009577273,0.05755316,0.01216863,0.8686191,0.05529565,0.00007416354],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09354493,0.003009463,0.8377293,0.004012758,0.0005295525,0.001166178,0.0370198,0.008719591,0.01426841],"genre_scores_gemma":[0.4476794,0.003144794,0.4916075,0.0006271696,0.0003586215,0.0008765376,0.04282805,0.001349478,0.0115283],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009535979,"threshold_uncertainty_score":0.031901,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4327724188","doi":"10.1016/j.datak.2023.102177","title":"Preface","year":2023,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Geography","authors":[{"name":"Aditya Ghose","is_ca":false},{"name":"Jennifer Horkoff","is_ca":false},{"name":"Vítor E. Silva Souza","is_ca":false},{"name":"Jeff Parsons","is_ca":true},{"name":"Jöerg Evermann","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1044293387069667,"gpt":0.331910048881099,"spread":0.2274807101741324,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002034063,0.0009755931,0.0006473601,0.00399603,0.002413294,0.003709936,0.001184332,0.0008443161,0.4839282],"category_scores_gemma":[0.01971979,0.0003381301,0.0005571226,0.002760882,0.0005981335,0.002957699,0.002159719,0.002990698,0.3525421],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001856712,"about_ca_system_score_gemma":0.002505992,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005792624,"about_ca_topic_score_gemma":0.006536598,"domain_scores_codex":[0.9990485,0.0001399642,0.00006236473,0.0001400148,0.0005415017,0.00006753358],"domain_scores_gemma":[0.9879357,0.001899104,0.0003285655,0.00110206,0.007667259,0.001067332],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001603743,0.00001279614,0.00005990958,0.00006301624,0.000001128537,0.00001962533,0.00003198967,0.00004782985,0.00009073663,0.00329269,0.9652375,0.03112664],"study_design_scores_gemma":[0.000003316786,0.000008015373,0.0001629351,0.00009549064,0.000001493505,0.00002549307,0.00004945917,0.00003462894,0.0001003274,0.003029274,0.9964851,0.000004401165],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"editorial","genre_scores_codex":[0.001353405,0.007532759,0.02244202,0.03760052,0.2295315,0.001164893,0.02675934,0.003846845,0.6697687],"genre_scores_gemma":[0.004360482,0.003530684,0.006362096,0.006325968,0.02492838,0.0004107586,0.01507188,0.001694633,0.937315],"genre_candidate":"editorial","genre_consensus":null,"teacher_disagreement_score":0.5160718,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4412980931","doi":"10.1016/j.datak.2025.102499","title":"Corrigendum to “Large Language Models for Conceptual Modeling: Assessment and Application Potential” [Knowledge and Data Engineering Volume 160, November 2025, 102480]","year":2025,"lang":"en","type":"erratum","venue":"Data & Knowledge Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Volume (thermodynamics); Computer science; Conceptual model; Data science; Database; Physics","authors":[{"name":"Veda C. Storey","is_ca":false},{"name":"Jeffrey Parsons","is_ca":true},{"name":"Arturo Castellanos Bueso","is_ca":false},{"name":"Monica Chiarini Tremblay","is_ca":false},{"name":"Roman Lukyanenko","is_ca":false},{"name":"Alfred Castillo","is_ca":false},{"name":"Wolfgang Maaß","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04627239330600777,"gpt":0.3163658736183723,"spread":0.2700934803123646,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004909043,0.002469033,0.002380468,0.006057404,0.005960922,0.008802189,0.003707689,0.007981366,0.2137453],"category_scores_gemma":[0.04102559,0.001357388,0.002527323,0.004284673,0.001775109,0.004525512,0.002904458,0.007878931,0.1587043],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008395775,"about_ca_system_score_gemma":0.006837992,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.08813502,"about_ca_topic_score_gemma":0.1342151,"domain_scores_codex":[0.9929202,0.001170084,0.0006864503,0.0008516067,0.00386456,0.0005071941],"domain_scores_gemma":[0.9663382,0.0053836,0.000722807,0.001839002,0.02476387,0.0009524017],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000007958095,0.000003677651,0.00001135064,0.0000188757,0.000002200527,0.00001412753,0.000003844241,0.00002608441,0.00001786471,0.0005266033,0.9973673,0.002000152],"study_design_scores_gemma":[0.00002497611,0.00001932901,0.0004693245,0.000152109,0.00002675789,0.00005684229,0.00005060448,0.0008098671,0.0002824118,0.003641827,0.9944285,0.00003737384],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"other","genre_scores_codex":[0.0002204591,0.001766865,0.004826576,0.1080756,0.834114,0.0001333892,0.004434102,0.002054675,0.0443744],"genre_scores_gemma":[0.004030976,0.00353805,0.006850316,0.06389283,0.09925779,0.0003024086,0.008664257,0.002269028,0.8111943],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.2137453,"threshold_uncertainty_score":0.7150494,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}