{"meta":{"query_hash":"6ae6e644c581","filters":{"venue":"Knowledge Discovery and Data Mining"},"cohort_total":6,"direct_labels_cover":0,"predictions_cover":6,"exported":6,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/6ae6e644c581","api":"https://metacan.xera.ac/api/v1/cohort?venue=Knowledge+Discovery+and+Data+Mining"},"results":[{"id":"W2415000644","doi":"","title":"Learning Nonstationary Models of Normal Network Traffic for Detecting Novel Attacks, Edmonton, Alberta","year":2002,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Computer network; Artificial intelligence; Computer security","score_opus":0.058735039018205946,"score_gpt":0.2885419494280673,"score_spread":0.22980691040986137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2415000644","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29180723,0.010022275,0.6610679,0.0073887217,0.00093604706,0.00015615846,0.0019643032,0.00491129,0.021746038],"genre_scores_gemma":[0.67566234,0.009984564,0.20114644,0.00041059754,0.00028211533,0.00007308626,0.0040933257,0.00032603898,0.10802149],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966383,0.00005170625,0.000017150049,0.00007614285,0.00014774298,0.00004339842],"domain_scores_gemma":[0.9986111,0.00062921987,0.000036385278,0.00009198441,0.00055240834,0.00007885263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015417167,0.0006346593,0.00061215047,0.0014093408,0.00079158804,0.0012394678,0.00092458556,0.0007002345,0.002090765],"category_scores_gemma":[0.0036949096,0.00046684916,0.00037289746,0.0010820192,0.0005918025,0.00080842094,0.0003681728,0.00082005165,0.000650239],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006283021,0.00028488023,0.0147634465,0.0001483309,0.00013626495,0.00026231637,0.000277906,0.43484193,0.0064754006,0.0079021985,0.03249653,0.50178254],"study_design_scores_gemma":[0.000018021266,0.000034132205,0.0039965454,0.000010819097,0.000030503354,0.00003746631,0.00009141948,0.9856188,0.0022430865,0.00227826,0.0056273537,0.000013454573],"about_ca_topic_score_codex":0.57223517,"about_ca_topic_score_gemma":0.6311201,"teacher_disagreement_score":0.57223517,"about_ca_system_score_codex":0.0028052174,"about_ca_system_score_gemma":0.0043487777,"threshold_uncertainty_score":0.86056805},"labels":[],"label_agreement":null},{"id":"W2805055069","doi":"","title":"Ensemble-Based Anomaly Detetction using Cooperative Learning.","year":2017,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Ensemble learning; Anomaly detection; Artificial intelligence; Anomaly (physics); Machine learning; Physics","score_opus":0.08496144598796657,"score_gpt":0.35238607384059706,"score_spread":0.2674246278526305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805055069","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05075369,0.0003538658,0.94485116,0.00015538475,0.00011109488,0.00007013813,0.00015147019,0.002209058,0.0013441636],"genre_scores_gemma":[0.75392044,0.00016661115,0.24284884,0.000111893285,0.00010078144,0.00012080147,0.0007265972,0.00014322386,0.001860851],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982492,0.00040374912,0.00010369976,0.00044335978,0.00056829414,0.00023170086],"domain_scores_gemma":[0.9943461,0.0021579666,0.00047786444,0.0011003,0.0016365866,0.00028114126],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027510487,0.0008833846,0.0018187724,0.0022732064,0.0007521545,0.0010183719,0.0024811162,0.0011353478,0.0011603732],"category_scores_gemma":[0.007489802,0.0003955584,0.0010059965,0.0020325708,0.00042833746,0.0022168064,0.0020402323,0.0016505609,0.0005970172],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055475935,0.0007953213,0.017288845,0.00010627564,0.000493528,0.00033625367,0.0003114504,0.25564066,0.018826727,0.004106817,0.0077807894,0.69375855],"study_design_scores_gemma":[0.000006340867,0.000044178767,0.00077592675,0.0000035021264,0.00003170279,0.00005945408,0.000028927065,0.9943212,0.0023166372,0.0019833616,0.00042113688,0.0000076165693],"about_ca_topic_score_codex":0.0044160276,"about_ca_topic_score_gemma":0.0054841307,"teacher_disagreement_score":0.0044160276,"about_ca_system_score_codex":0.000477255,"about_ca_system_score_gemma":0.0009480141,"threshold_uncertainty_score":0.014549136},"labels":[],"label_agreement":null},{"id":"W3003238448","doi":"","title":"Towards a Benchmark for Knowledge Base Exchange.","year":2019,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Benchmark (surveying); Computer science; Knowledge base; Base (topology); Artificial intelligence; Mathematics; Geology","score_opus":0.06517942266695431,"score_gpt":0.32064177626958146,"score_spread":0.25546235360262715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3003238448","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.071025535,0.008936729,0.72474587,0.016950205,0.003259316,0.0031608066,0.03171396,0.051414758,0.08879288],"genre_scores_gemma":[0.1789084,0.0025964912,0.70590794,0.0013639228,0.00040022907,0.0014747416,0.097580604,0.0034502426,0.008317432],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9581049,0.016803006,0.007095715,0.0035562825,0.012393637,0.0020465304],"domain_scores_gemma":[0.89087594,0.035765328,0.0036759807,0.034831118,0.02905588,0.0057956795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05331379,0.0016210134,0.0021829396,0.010430178,0.0042434,0.018003548,0.008788578,0.0056799925,0.009483195],"category_scores_gemma":[0.183088,0.0011138597,0.0015162908,0.013540904,0.0022196146,0.027204506,0.010828358,0.0052292086,0.007816269],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002170003,0.0025911715,0.010660645,0.0022654743,0.0004280249,0.0005773092,0.001666745,0.025213758,0.0069778655,0.21296802,0.16807695,0.56640404],"study_design_scores_gemma":[0.0006535456,0.0011696399,0.007126744,0.0024621382,0.00035906088,0.0010838038,0.0031440828,0.2142626,0.037108585,0.4271602,0.3052155,0.00025404766],"about_ca_topic_score_codex":0.00974718,"about_ca_topic_score_gemma":0.007532618,"teacher_disagreement_score":0.05331379,"about_ca_system_score_codex":0.003710047,"about_ca_system_score_gemma":0.009804478,"threshold_uncertainty_score":0.28195363},"labels":[],"label_agreement":null},{"id":"W3004067286","doi":"","title":"Detecting and Correcting Typing Errors in DBpedia.","year":2019,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Typing; Information retrieval; Artificial intelligence; Speech recognition","score_opus":0.03783106692397872,"score_gpt":0.3066728428562663,"score_spread":0.26884177593228753,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004067286","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17467025,0.0071981153,0.59210545,0.006023504,0.0058962065,0.0015230237,0.09733302,0.09329607,0.021954427],"genre_scores_gemma":[0.24770828,0.0023263479,0.61867094,0.0022107353,0.000493262,0.00042621876,0.11094508,0.008071883,0.009147318],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9836586,0.0044754134,0.0023831974,0.0033037493,0.0053934716,0.0007855901],"domain_scores_gemma":[0.9362258,0.029126272,0.0044845985,0.014039313,0.015169177,0.00095496455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0095052,0.0016865684,0.001533593,0.008312087,0.0026687917,0.00629591,0.0031526107,0.0026400993,0.0022337746],"category_scores_gemma":[0.06935366,0.0011655254,0.0014235914,0.0067532524,0.001005926,0.0064822594,0.0049691177,0.0030078043,0.004261702],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015028688,0.0012854362,0.06061647,0.00450612,0.0011086265,0.004624105,0.005694993,0.008762404,0.030614763,0.0165613,0.29534572,0.5693772],"study_design_scores_gemma":[0.0003020753,0.00041544708,0.029871752,0.0036178066,0.0014887219,0.005351347,0.0071592345,0.14825702,0.18153411,0.07943864,0.54189223,0.00067164155],"about_ca_topic_score_codex":0.008348448,"about_ca_topic_score_gemma":0.012223271,"teacher_disagreement_score":0.0095052,"about_ca_system_score_codex":0.0009587288,"about_ca_system_score_gemma":0.0051576677,"threshold_uncertainty_score":0.05026889},"labels":[],"label_agreement":null},{"id":"W62188274","doi":"10.5555/1760894.1760939","title":"Position coded pre-order linked WAP-tree for web log sequential pattern mining","year":2003,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Prefix; Computer science; Tree (set theory); Trie; Fractal tree index; Web log analysis software; Suffix tree; Preorder; Suffix; Node (physics); Segment tree; Binary tree; Tree structure; Data mining; Interval tree; Data structure; Algorithm; Web server; Mathematics; World Wide Web; The Internet; Combinatorics; Discrete mathematics; Operating system","score_opus":0.0458814739950354,"score_gpt":0.3150776749332564,"score_spread":0.269196200938221,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W62188274","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07068932,0.00028229764,0.91159624,0.0002493764,0.00015846733,0.0003560605,0.004157714,0.009114987,0.003395488],"genre_scores_gemma":[0.30465794,0.00015300456,0.6833202,0.00010162737,0.000044712353,0.00036152318,0.0067467424,0.00033955774,0.0042746365],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994723,0.00008494771,0.0000660927,0.000089997586,0.00021933482,0.000067329194],"domain_scores_gemma":[0.997686,0.0007625698,0.000111289286,0.0006402308,0.0006874198,0.000112405425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068259786,0.0003256922,0.000583097,0.0020634297,0.00089054496,0.0011508848,0.0014009298,0.0008913896,0.0044374163],"category_scores_gemma":[0.0060026087,0.00031098444,0.00050084665,0.0028287114,0.00037452826,0.0013433556,0.0007991502,0.00096869835,0.0015443166],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015050324,0.00051534513,0.0077465116,0.00038683828,0.000073421295,0.0006871356,0.00031700506,0.0726266,0.018625591,0.02821346,0.018453307,0.8508497],"study_design_scores_gemma":[0.0000984909,0.00024716018,0.0021356973,0.00006914465,0.000059712915,0.00034145828,0.0001510062,0.9245768,0.015539753,0.0432299,0.013506464,0.00004440171],"about_ca_topic_score_codex":0.0050740107,"about_ca_topic_score_gemma":0.008183492,"teacher_disagreement_score":0.0050740107,"about_ca_system_score_codex":0.00049100525,"about_ca_system_score_gemma":0.0018199207,"threshold_uncertainty_score":0.014844656},"labels":[],"label_agreement":null},{"id":"W73242358","doi":"","title":"Matching Unstructured Offers to Structured Product Descriptions","year":2011,"lang":"en","type":"article","venue":"Knowledge Discovery and Data Mining","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Matching (statistics); Product (mathematics); Component (thermodynamics); Function (biology); Information retrieval; Database; World Wide Web","score_opus":0.07683517453153065,"score_gpt":0.2881091713015354,"score_spread":0.21127399677000475,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W73242358","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4234449,0.0007813415,0.5049238,0.0013994576,0.00019714097,0.001260351,0.02100552,0.032007746,0.014979691],"genre_scores_gemma":[0.44660598,0.00032360226,0.51455426,0.00044581454,0.00005846481,0.00021567314,0.030802162,0.0007748322,0.006219278],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99824715,0.0002784183,0.0002012952,0.0004721291,0.0007158299,0.00008514592],"domain_scores_gemma":[0.99482524,0.0026286484,0.00060833374,0.0010158753,0.0007181194,0.00020384627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001953774,0.0006226811,0.00072205736,0.0035682772,0.0005247964,0.002077727,0.0015759884,0.001025189,0.006714671],"category_scores_gemma":[0.014228028,0.00047440952,0.0007366433,0.0031818408,0.00046466212,0.0036959245,0.002048856,0.00074175897,0.0024078144],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019016948,0.0014585229,0.04512915,0.0011011367,0.0003121494,0.001479739,0.0015179378,0.04679032,0.025603686,0.031105718,0.05311886,0.79048115],"study_design_scores_gemma":[0.00020715664,0.0004391403,0.01305419,0.00014978931,0.00017382725,0.0010604417,0.0013126909,0.8041967,0.053906735,0.04545854,0.079896614,0.00014418401],"about_ca_topic_score_codex":0.0062731234,"about_ca_topic_score_gemma":0.009098802,"teacher_disagreement_score":0.006714671,"about_ca_system_score_codex":0.0009208868,"about_ca_system_score_gemma":0.0015393874,"threshold_uncertainty_score":0.022462845},"labels":[],"label_agreement":null}]}