{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":6,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":6,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"6ae6e644c581","filters":{"venue":"Knowledge Discovery and Data Mining"}},"results":[{"id":"W62188274","doi":"10.5555/1760894.1760939","title":"Position coded pre-order linked WAP-tree for web log sequential pattern mining","year":2003,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Windsor","funders":"","keywords":"Prefix; Computer science; Tree (set theory); Trie; Fractal tree index; Web log analysis software; Suffix tree; Preorder; Suffix; Node (physics); Segment tree; Binary tree; Tree structure; Data mining; Interval tree; Data structure; Algorithm; Web server; Mathematics; World Wide Web; The Internet; Combinatorics; Discrete mathematics; Operating system","authors":[{"name":"Yi Lu","is_ca":true},{"name":"C. I. Ezeife","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0458814739950354,"gpt":0.3150776749332564,"spread":0.269196200938221,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006825979,0.0003256922,0.000583097,0.00206343,0.000890545,0.001150885,0.00140093,0.0008913896,0.004437416],"category_scores_gemma":[0.006002609,0.0003109844,0.0005008467,0.002828711,0.0003745283,0.001343356,0.0007991502,0.0009686984,0.001544317],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004910053,"about_ca_system_score_gemma":0.001819921,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005074011,"about_ca_topic_score_gemma":0.008183492,"domain_scores_codex":[0.9994723,0.00008494771,0.0000660927,0.00008999759,0.0002193348,0.00006732919],"domain_scores_gemma":[0.997686,0.0007625698,0.0001112893,0.0006402308,0.0006874198,0.0001124054],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001505032,0.0005153451,0.007746512,0.0003868383,0.0000734213,0.0006871356,0.0003170051,0.0726266,0.01862559,0.02821346,0.01845331,0.8508497],"study_design_scores_gemma":[0.0000984909,0.0002471602,0.002135697,0.00006914465,0.00005971291,0.0003414583,0.0001510062,0.9245768,0.01553975,0.0432299,0.01350646,0.00004440171],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07068932,0.0002822976,0.9115962,0.0002493764,0.0001584673,0.0003560605,0.004157714,0.009114987,0.003395488],"genre_scores_gemma":[0.3046579,0.0001530046,0.6833202,0.0001016274,0.00004471235,0.0003615232,0.006746742,0.0003395577,0.004274637],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005074011,"threshold_uncertainty_score":0.01484466,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2805055069","doi":"","title":"Ensemble-Based Anomaly Detetction using Cooperative Learning.","year":2017,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Ensemble learning; Anomaly detection; Artificial intelligence; Anomaly (physics); Machine learning; Physics","authors":[{"name":"Rasha Kashef","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08496144598796657,"gpt":0.3523860738405971,"spread":0.2674246278526305,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002751049,0.0008833846,0.001818772,0.002273206,0.0007521545,0.001018372,0.002481116,0.001135348,0.001160373],"category_scores_gemma":[0.007489802,0.0003955584,0.001005996,0.002032571,0.0004283375,0.002216806,0.002040232,0.001650561,0.0005970172],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000477255,"about_ca_system_score_gemma":0.0009480141,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004416028,"about_ca_topic_score_gemma":0.005484131,"domain_scores_codex":[0.9982492,0.0004037491,0.0001036998,0.0004433598,0.0005682941,0.0002317009],"domain_scores_gemma":[0.9943461,0.002157967,0.0004778644,0.0011003,0.001636587,0.0002811413],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005547593,0.0007953213,0.01728885,0.0001062756,0.000493528,0.0003362537,0.0003114504,0.2556407,0.01882673,0.004106817,0.007780789,0.6937585],"study_design_scores_gemma":[0.000006340867,0.00004417877,0.0007759267,0.000003502126,0.00003170279,0.00005945408,0.00002892706,0.9943212,0.002316637,0.001983362,0.0004211369,0.000007616569],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05075369,0.0003538658,0.9448512,0.0001553848,0.0001110949,0.00007013813,0.0001514702,0.002209058,0.001344164],"genre_scores_gemma":[0.7539204,0.0001666111,0.2428488,0.0001118933,0.0001007814,0.0001208015,0.0007265972,0.0001432239,0.001860851],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004416028,"threshold_uncertainty_score":0.01454914,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2415000644","doi":"","title":"Learning Nonstationary Models of Normal Network Traffic for Detecting Novel Attacks, Edmonton, Alberta","year":2002,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Computer science; Computer network; Artificial intelligence; Computer security","authors":[{"name":"Matthew V. Mahoney","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05873503901820595,"gpt":0.2885419494280673,"spread":0.2298069104098614,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001541717,0.0006346593,0.0006121505,0.001409341,0.000791588,0.001239468,0.0009245856,0.0007002345,0.002090765],"category_scores_gemma":[0.00369491,0.0004668492,0.0003728975,0.001082019,0.0005918025,0.0008084209,0.0003681728,0.0008200516,0.000650239],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002805217,"about_ca_system_score_gemma":0.004348778,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.5722352,"about_ca_topic_score_gemma":0.6311201,"domain_scores_codex":[0.9996638,0.00005170625,0.00001715005,0.00007614285,0.000147743,0.00004339842],"domain_scores_gemma":[0.9986111,0.0006292199,0.00003638528,0.00009198441,0.0005524083,0.00007885263],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006283021,0.0002848802,0.01476345,0.0001483309,0.0001362649,0.0002623164,0.000277906,0.4348419,0.006475401,0.007902198,0.03249653,0.5017825],"study_design_scores_gemma":[0.00001802127,0.0000341322,0.003996545,0.0000108191,0.00003050335,0.00003746631,0.00009141948,0.9856188,0.002243086,0.00227826,0.005627354,0.00001345457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2918072,0.01002228,0.6610679,0.007388722,0.0009360471,0.0001561585,0.001964303,0.00491129,0.02174604],"genre_scores_gemma":[0.6756623,0.009984564,0.2011464,0.0004105975,0.0002821153,0.00007308626,0.004093326,0.000326039,0.1080215],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5722352,"threshold_uncertainty_score":0.860568,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W73242358","doi":"","title":"Matching Unstructured Offers to Structured Product Descriptions","year":2011,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Matching (statistics); Product (mathematics); Component (thermodynamics); Function (biology); Information retrieval; Database; World Wide Web","authors":[{"name":"Anitha Kannan","is_ca":false},{"name":"Inmar E. Givoni","is_ca":true},{"name":"Rakesh Agrawal","is_ca":false},{"name":"Ariel Fuxman","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07683517453153065,"gpt":0.2881091713015354,"spread":0.2112739967700047,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001953774,0.0006226811,0.0007220574,0.003568277,0.0005247964,0.002077727,0.001575988,0.001025189,0.006714671],"category_scores_gemma":[0.01422803,0.0004744095,0.0007366433,0.003181841,0.0004646621,0.003695925,0.002048856,0.000741759,0.002407814],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009208868,"about_ca_system_score_gemma":0.001539387,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006273123,"about_ca_topic_score_gemma":0.009098802,"domain_scores_codex":[0.9982471,0.0002784183,0.0002012952,0.0004721291,0.0007158299,0.00008514592],"domain_scores_gemma":[0.9948252,0.002628648,0.0006083337,0.001015875,0.0007181194,0.0002038463],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001901695,0.001458523,0.04512915,0.001101137,0.0003121494,0.001479739,0.001517938,0.04679032,0.02560369,0.03110572,0.05311886,0.7904812],"study_design_scores_gemma":[0.0002071566,0.0004391403,0.01305419,0.0001497893,0.0001738273,0.001060442,0.001312691,0.8041967,0.05390674,0.04545854,0.07989661,0.000144184],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4234449,0.0007813415,0.5049238,0.001399458,0.000197141,0.001260351,0.02100552,0.03200775,0.01497969],"genre_scores_gemma":[0.446606,0.0003236023,0.5145543,0.0004458145,0.00005846481,0.0002156731,0.03080216,0.0007748322,0.006219278],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006714671,"threshold_uncertainty_score":0.02246284,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3003238448","doi":"","title":"Towards a Benchmark for Knowledge Base Exchange.","year":2019,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Benchmark (surveying); Computer science; Knowledge base; Base (topology); Artificial intelligence; Mathematics; Geology","authors":[{"name":"Bahar Ghadiri Bashardoost","is_ca":true},{"name":"Renée J. Miller","is_ca":false},{"name":"Kelly Lyons","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06517942266695431,"gpt":0.3206417762695815,"spread":0.2554623536026271,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05331379,0.001621013,0.00218294,0.01043018,0.0042434,0.01800355,0.008788578,0.005679992,0.009483195],"category_scores_gemma":[0.183088,0.00111386,0.001516291,0.0135409,0.002219615,0.02720451,0.01082836,0.005229209,0.007816269],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003710047,"about_ca_system_score_gemma":0.009804478,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00974718,"about_ca_topic_score_gemma":0.007532618,"domain_scores_codex":[0.9581049,0.01680301,0.007095715,0.003556282,0.01239364,0.00204653],"domain_scores_gemma":[0.8908759,0.03576533,0.003675981,0.03483112,0.02905588,0.00579568],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002170003,0.002591172,0.01066064,0.002265474,0.0004280249,0.0005773092,0.001666745,0.02521376,0.006977865,0.212968,0.1680769,0.566404],"study_design_scores_gemma":[0.0006535456,0.00116964,0.007126744,0.002462138,0.0003590609,0.001083804,0.003144083,0.2142626,0.03710859,0.4271602,0.3052155,0.0002540477],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07102554,0.008936729,0.7247459,0.0169502,0.003259316,0.003160807,0.03171396,0.05141476,0.08879288],"genre_scores_gemma":[0.1789084,0.002596491,0.7059079,0.001363923,0.0004002291,0.001474742,0.0975806,0.003450243,0.008317432],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.05331379,"threshold_uncertainty_score":0.2819536,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3004067286","doi":"","title":"Detecting and Correcting Typing Errors in DBpedia.","year":2019,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Typing; Information retrieval; Artificial intelligence; Speech recognition","authors":[{"name":"Daniel D. Caminhas","is_ca":false},{"name":"Daniel Cones","is_ca":false},{"name":"Natalie Hervieux","is_ca":false},{"name":"Denilson Barbosa","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03783106692397872,"gpt":0.3066728428562663,"spread":0.2688417759322875,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0095052,0.001686568,0.001533593,0.008312087,0.002668792,0.00629591,0.003152611,0.002640099,0.002233775],"category_scores_gemma":[0.06935366,0.001165525,0.001423591,0.006753252,0.001005926,0.006482259,0.004969118,0.003007804,0.004261702],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009587288,"about_ca_system_score_gemma":0.005157668,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008348448,"about_ca_topic_score_gemma":0.01222327,"domain_scores_codex":[0.9836586,0.004475413,0.002383197,0.003303749,0.005393472,0.0007855901],"domain_scores_gemma":[0.9362258,0.02912627,0.004484599,0.01403931,0.01516918,0.0009549645],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001502869,0.001285436,0.06061647,0.00450612,0.001108626,0.004624105,0.005694993,0.008762404,0.03061476,0.0165613,0.2953457,0.5693772],"study_design_scores_gemma":[0.0003020753,0.0004154471,0.02987175,0.003617807,0.001488722,0.005351347,0.007159234,0.148257,0.1815341,0.07943864,0.5418922,0.0006716416],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1746702,0.007198115,0.5921054,0.006023504,0.005896206,0.001523024,0.09733302,0.09329607,0.02195443],"genre_scores_gemma":[0.2477083,0.002326348,0.6186709,0.002210735,0.000493262,0.0004262188,0.1109451,0.008071883,0.009147318],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0095052,"threshold_uncertainty_score":0.05026889,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}