{"meta":{"query_hash":"81a69d3f0a61","filters":{"venue":"Data Science and Engineering"},"cohort_total":4,"direct_labels_cover":0,"predictions_cover":4,"exported":4,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/81a69d3f0a61","api":"https://metacan.xera.ac/api/v1/cohort?venue=Data+Science+and+Engineering"},"results":[{"id":"W3155426438","doi":"10.1007/s41019-021-00156-2","title":"A Workload-Adaptive Streaming Partitioner for Distributed Graph Stores","year":2021,"lang":"en","type":"article","venue":"Data Science and Engineering","topic":"Graph Theory and Algorithms","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Exploit; Workload; Graph traversal; Graph; Tree traversal; Latency (audio); Theoretical computer science; Distributed computing; Operating system; Algorithm","score_opus":0.026221650197992386,"score_gpt":0.23748156179220617,"score_spread":0.21125991159421378,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3155426438","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12668863,0.0006995634,0.80182683,0.00065571757,0.00017435047,0.0005235078,0.0018645866,0.06511262,0.002454259],"genre_scores_gemma":[0.4443191,0.0002725469,0.5458831,0.00024852162,0.000096380405,0.00038034655,0.0036696065,0.0017811767,0.0033491843],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990709,0.00013324291,0.00008250265,0.00023386483,0.0003977276,0.000081760896],"domain_scores_gemma":[0.9973978,0.0008003916,0.00016759496,0.0009416493,0.00043834466,0.00025409853],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010066149,0.00093642547,0.00094932923,0.001314389,0.0007698287,0.0014019923,0.0033549534,0.0008396744,0.0033476301],"category_scores_gemma":[0.0044226698,0.0005565913,0.0005146963,0.0018029631,0.0006040291,0.0031465504,0.002470179,0.0011609155,0.0010604034],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030477082,0.0010091563,0.007400671,0.00072179164,0.00021109417,0.00063193415,0.0011044528,0.123514004,0.1542224,0.01387947,0.04524042,0.64901686],"study_design_scores_gemma":[0.00018460899,0.00016493854,0.0007182073,0.000014910173,0.000025033029,0.00016963114,0.00018320451,0.9555231,0.030246634,0.0053698393,0.007364026,0.00003589627],"about_ca_topic_score_codex":0.0032026097,"about_ca_topic_score_gemma":0.0044745803,"teacher_disagreement_score":0.0033549534,"about_ca_system_score_codex":0.00086225604,"about_ca_system_score_gemma":0.0010171133,"threshold_uncertainty_score":0.011198938},"labels":[],"label_agreement":null},{"id":"W4243973573","doi":"10.1007/s41019-017-0053-1","title":"Sliding Window Top-K Monitoring over Distributed Data Streams","year":2017,"lang":"en","type":"article","venue":"Data Science and Engineering","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"National Key Research and Development Program of China; Natural Science Foundation of Shandong Province; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Computer science; Scalability; Sliding window protocol; Distributed computing; Overhead (engineering); Data stream mining; Distributed Computing Environment; Distributed algorithm; Aggregate (composite); Window (computing); Data stream; Real-time computing; Data mining; Database; Operating system","score_opus":0.06100806266574163,"score_gpt":0.2964756951502873,"score_spread":0.2354676324845457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4243973573","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09881251,0.00057500444,0.8962637,0.00014122113,0.000059332913,0.0001098127,0.00012379565,0.0033097586,0.0006048605],"genre_scores_gemma":[0.7732824,0.00021864836,0.2254398,0.000045409954,0.00004597057,0.00007658989,0.00020964316,0.000112694244,0.00056875014],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99788314,0.00032634588,0.00020346639,0.00072302506,0.00063095277,0.0002331495],"domain_scores_gemma":[0.9954364,0.0017340536,0.0009170209,0.0009432589,0.00064413506,0.000325169],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021727206,0.0010718242,0.0019589802,0.0013900369,0.0011685381,0.0019220245,0.001943457,0.0006219027,0.00055904186],"category_scores_gemma":[0.007877362,0.0005326995,0.0005850716,0.0020967626,0.0008598796,0.0031113606,0.0016742336,0.00095898454,0.00020892113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021268958,0.00034625948,0.020748958,0.0002820256,0.000259211,0.00051632844,0.00082104775,0.3916993,0.04859462,0.007049248,0.00322003,0.52433604],"study_design_scores_gemma":[0.000028201059,0.00007500013,0.0010281974,0.000004944734,0.000025888916,0.00008770394,0.00006926727,0.9873713,0.007629842,0.0030673256,0.00060011004,0.000012258576],"about_ca_topic_score_codex":0.0050561493,"about_ca_topic_score_gemma":0.00408344,"teacher_disagreement_score":0.0050561493,"about_ca_system_score_codex":0.00095830596,"about_ca_system_score_gemma":0.0013831807,"threshold_uncertainty_score":0.011490583},"labels":[],"label_agreement":null},{"id":"W4292512861","doi":"10.1007/s41019-022-00193-5","title":"Dimensionality Reduction in Surrogate Modeling: A Review of Combined Methods","year":2022,"lang":"en","type":"review","venue":"Data Science and Engineering","topic":"Advanced Multi-Objective Optimization Algorithms","field":"Computer Science","cited_by":117,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Surrogate model; Dimensionality reduction; Curse of dimensionality; Computer science; Uncertainty quantification; Reduction (mathematics); Artificial intelligence; Mathematical optimization; Machine learning; Mathematics","score_opus":0.15616597191542345,"score_gpt":0.42200037778948857,"score_spread":0.26583440587406515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4292512861","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00048512343,0.94731724,0.04816126,0.0005989266,0.00043472878,0.000037725174,0.000082158156,0.000076847806,0.0028060062],"genre_scores_gemma":[0.0060553285,0.9654381,0.02622193,0.0002548196,0.0006937223,0.000099106794,0.00016421305,0.000051075487,0.0010216461],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984281,0.0004547194,0.00023187902,0.00023929875,0.0005936267,0.000052408883],"domain_scores_gemma":[0.9968844,0.0021179952,0.00018675507,0.00014485158,0.0006106627,0.00005515249],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028396337,0.0016984879,0.002705367,0.0036334048,0.0004031257,0.0023673477,0.0016029258,0.0016554751,0.0038275952],"category_scores_gemma":[0.005493491,0.0008085365,0.0020019866,0.005649243,0.000922846,0.0023784405,0.0013355649,0.002107615,0.001796935],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000064211075,0.00010281147,0.00044635677,0.016695464,0.00030068975,0.00013667722,0.00011066516,0.013879827,0.0014195136,0.03512606,0.015240097,0.91647774],"study_design_scores_gemma":[0.000056506637,0.00040457622,0.001679634,0.0119948145,0.0007122155,0.001689805,0.0001716408,0.04744591,0.004852943,0.055338494,0.8753825,0.0002710152],"about_ca_topic_score_codex":0.0012900063,"about_ca_topic_score_gemma":0.0010933747,"teacher_disagreement_score":0.0038275952,"about_ca_system_score_codex":0.0007311967,"about_ca_system_score_gemma":0.0014183761,"threshold_uncertainty_score":0.015017629},"labels":[],"label_agreement":null},{"id":"W4393347796","doi":"10.1007/s41019-023-00239-2","title":"Uncovering Flat and Hierarchical Topics by Community Discovery on Word Co-occurrence Network","year":2024,"lang":"en","type":"article","venue":"Data Science and Engineering","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke; University of Alberta","funders":"Centre National de la Recherche Scientifique; Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Hierarchy; Identification (biology); Topic model; Word (group theory); Exploit; Data science; Tree (set theory); Information retrieval; Linguistics","score_opus":0.023266520834801834,"score_gpt":0.2920903359651108,"score_spread":0.26882381513030895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393347796","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10570736,0.0025527207,0.88523805,0.00045308008,0.00007654077,0.0003280116,0.0016259494,0.001611414,0.0024068758],"genre_scores_gemma":[0.5231264,0.0014504587,0.4648589,0.00012546014,0.00024592894,0.00043918742,0.0065794517,0.00027951936,0.0028946784],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981628,0.00049709855,0.000100170044,0.00061034964,0.0004274637,0.00020216775],"domain_scores_gemma":[0.9953041,0.002745843,0.00067224674,0.00036066226,0.0006919269,0.00022513827],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001791662,0.0013981748,0.0012200157,0.011115947,0.0015237153,0.0016083416,0.0013337167,0.0012995952,0.00083327875],"category_scores_gemma":[0.0068435282,0.000492062,0.0014540373,0.006835887,0.0008192831,0.0039798818,0.0021311024,0.0012315759,0.00076365203],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010034158,0.00056594465,0.052075822,0.0011625905,0.00065513887,0.0012323642,0.0034875884,0.11048073,0.025864284,0.028577656,0.01928394,0.7556105],"study_design_scores_gemma":[0.000041220825,0.00006823376,0.005045797,0.000075394186,0.0001326399,0.00044302494,0.0006327314,0.9463732,0.0050706523,0.03491358,0.0071538663,0.00004963932],"about_ca_topic_score_codex":0.006110988,"about_ca_topic_score_gemma":0.008862048,"teacher_disagreement_score":0.011115947,"about_ca_system_score_codex":0.00077474845,"about_ca_system_score_gemma":0.0012281577,"threshold_uncertainty_score":0.012150824},"labels":[],"label_agreement":null}]}