{"meta":{"query_hash":"f35200098bd0","filters":{"venue":"Statistical Analysis and Data Mining The ASA Data Science Journal"},"cohort_total":37,"direct_labels_cover":1,"predictions_cover":37,"exported":37,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/f35200098bd0","api":"https://metacan.xera.ac/api/v1/cohort?venue=Statistical+Analysis+and+Data+Mining+The+ASA+Data+Science+Journal"},"results":[{"id":"W1512585278","doi":"10.1002/sam.11261","title":"Prediction using hierarchical data: Applications for automated detection of cervical cancer","year":2015,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"AI in cancer detection","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Occupational Cancer Research Centre","funders":"National Cancer Institute; National Institutes of Health","keywords":"Computer science; Papanicolaou stain; Cervical cancer; Artificial intelligence; Data mining; Pattern recognition (psychology); Machine learning; Cancer; Medicine","score_opus":0.18075572576648957,"score_gpt":0.4095357243360112,"score_spread":0.22877999856952164,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1512585278","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.115064,0.002567228,0.8645193,0.002704868,0.00020094345,0.000365541,0.0053955265,0.0070997765,0.0020827362],"genre_scores_gemma":[0.6137854,0.00082441093,0.3792456,0.00031573197,0.00017571288,0.00030352152,0.0042834166,0.00013353972,0.00093270175],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99772495,0.0010806568,0.00018357337,0.0003771631,0.0005235133,0.00011010144],"domain_scores_gemma":[0.986356,0.010316584,0.0007876122,0.0012095248,0.0010503845,0.0002799168],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048415507,0.0010272479,0.001024126,0.0031490824,0.00056709186,0.0009972237,0.001164626,0.0010418702,0.001789666],"category_scores_gemma":[0.022138039,0.00048044854,0.0011150185,0.0032573275,0.00049200293,0.0011570653,0.0012749126,0.0014531687,0.00067367026],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005043037,0.0004925677,0.07304504,0.00038085217,0.00032586762,0.0003498711,0.00038147447,0.39789918,0.0036922437,0.006086726,0.01013125,0.50671065],"study_design_scores_gemma":[0.000030169977,0.00005613331,0.004954003,0.000025634321,0.00002190018,0.000042351367,0.00006237686,0.98011005,0.0010986145,0.011912436,0.0016669781,0.000019331781],"about_ca_topic_score_codex":0.019511811,"about_ca_topic_score_gemma":0.021207249,"teacher_disagreement_score":0.019511811,"about_ca_system_score_codex":0.0010202,"about_ca_system_score_gemma":0.0013861906,"threshold_uncertainty_score":0.038796484},"labels":[],"label_agreement":null},{"id":"W1525895928","doi":"10.1002/sam.11394","title":"Standardizing interestingness measures for association rules","year":2018,"lang":"en","type":"preprint","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph; McMaster University; Thompson Rivers University","funders":"Ontario Ministry of Research and Innovation; Natural Sciences and Engineering Research Council of Canada","keywords":"Measure (data warehouse); Lift (data mining); Computer science; Raw data; Association rule learning; Standardization; Association (psychology); Data mining; Value (mathematics); Machine learning; Psychology","score_opus":0.12011645933756977,"score_gpt":0.3973333708886629,"score_spread":0.2772169115510931,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1525895928","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12049691,0.002910347,0.86872673,0.0011252942,0.00026205342,0.00043252818,0.00096805143,0.0007440138,0.004334023],"genre_scores_gemma":[0.58556116,0.0008490371,0.4091456,0.00041385472,0.00045504596,0.0011005031,0.001713499,0.0003093128,0.00045200862],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9270636,0.03142687,0.009621611,0.010938618,0.019757807,0.0011914774],"domain_scores_gemma":[0.577021,0.3136332,0.028638726,0.05472247,0.023714712,0.0022699533],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07722302,0.0019437418,0.0030587527,0.012958147,0.0015925863,0.008040673,0.0032188133,0.0029100345,0.0013124344],"category_scores_gemma":[0.35292897,0.0009869298,0.0023849981,0.010109475,0.0065304176,0.011208287,0.0053651077,0.005090179,0.00041465712],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010301556,0.0006186695,0.059247326,0.0015680771,0.0022460595,0.00042872253,0.0018023731,0.15016076,0.010642861,0.39578864,0.004936362,0.37153006],"study_design_scores_gemma":[0.00015547365,0.00072807254,0.009950608,0.00042007022,0.00033637026,0.00063906895,0.0004844565,0.23339345,0.012096748,0.7362395,0.0053640893,0.0001920911],"about_ca_topic_score_codex":0.00045792753,"about_ca_topic_score_gemma":0.00049358263,"teacher_disagreement_score":0.07722302,"about_ca_system_score_codex":0.0025043993,"about_ca_system_score_gemma":0.0020103345,"threshold_uncertainty_score":0.40839922},"labels":[],"label_agreement":null},{"id":"W1979148915","doi":"10.1002/sam.11184","title":"Content‐boosted matrix factorization techniques for recommender systems","year":2013,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Recommender system; Matrix decomposition; Collaborative filtering; Matrix (chemical analysis); Factorization; Non-negative matrix factorization","score_opus":0.15790303555913265,"score_gpt":0.37882805816453147,"score_spread":0.2209250226053988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1979148915","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036287212,0.0015643139,0.9927235,0.00044658346,0.00011475054,0.00006547835,0.000121555284,0.0002342436,0.0011009042],"genre_scores_gemma":[0.23322916,0.0026305902,0.7573197,0.000424579,0.0006530735,0.00032179404,0.0006677139,0.00008768989,0.004665733],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965988,0.0017043143,0.00013503949,0.0005309918,0.0008933729,0.00013742597],"domain_scores_gemma":[0.99005264,0.006794681,0.00053068285,0.0008844348,0.0015933214,0.0001443509],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040772087,0.0014196965,0.0016887898,0.0019429706,0.0008647368,0.0012948964,0.0016770886,0.0018934086,0.004750344],"category_scores_gemma":[0.01608636,0.00086231896,0.0015392746,0.0024962896,0.0008884579,0.00254389,0.0011312492,0.0031405226,0.0021570516],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002602563,0.00033968204,0.002049008,0.00095242116,0.00055581494,0.00021552971,0.00041315512,0.42145786,0.0094986865,0.10072381,0.019934716,0.4435991],"study_design_scores_gemma":[0.000027509166,0.00006111778,0.00024720147,0.00003994076,0.000041403688,0.000052424584,0.000019470064,0.9489918,0.0010554503,0.04614091,0.0032998207,0.000022917071],"about_ca_topic_score_codex":0.007765443,"about_ca_topic_score_gemma":0.008585968,"teacher_disagreement_score":0.007765443,"about_ca_system_score_codex":0.0009879312,"about_ca_system_score_gemma":0.00092385773,"threshold_uncertainty_score":0.021562636},"labels":[],"label_agreement":null},{"id":"W1982925244","doi":"10.1002/sam.10094","title":"Special issue on the best papers of SDM'10","year":2010,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Citation; Library science; Computer science; Information retrieval","score_opus":0.05194367942912519,"score_gpt":0.3425082124392971,"score_spread":0.2905645330101719,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982925244","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013346015,0.044955537,0.015630016,0.04641785,0.79049426,0.00027127136,0.003501992,0.001420424,0.09597408],"genre_scores_gemma":[0.010435206,0.038281992,0.015575831,0.015777841,0.41266128,0.000334563,0.008466535,0.002651616,0.49581513],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99602354,0.00061613054,0.0003705468,0.0005594409,0.0021173966,0.00031294118],"domain_scores_gemma":[0.98478055,0.002701741,0.00080956594,0.0015945253,0.0071986564,0.0029149002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005369399,0.0018590993,0.0024450598,0.0073137158,0.0017743567,0.009073788,0.0024868217,0.0025563098,0.12161495],"category_scores_gemma":[0.016767241,0.0006702429,0.001547108,0.0072542047,0.00092798553,0.003927201,0.0032256073,0.0032366966,0.0722213],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049106355,0.000032731357,0.00013400262,0.0002552406,0.00001815018,0.000032884796,0.000009103866,0.0001602401,0.00021834426,0.001933864,0.94016963,0.056986734],"study_design_scores_gemma":[0.000021812553,0.000050057162,0.0004528336,0.0002407441,0.00002267318,0.00009610502,0.000021163964,0.00049752754,0.0002740622,0.0048456746,0.99346185,0.000015439915],"about_ca_topic_score_codex":0.0015769198,"about_ca_topic_score_gemma":0.0041284026,"teacher_disagreement_score":0.12161495,"about_ca_system_score_codex":0.0032265256,"about_ca_system_score_gemma":0.0045278193,"threshold_uncertainty_score":0.40684253},"labels":[],"label_agreement":null},{"id":"W2015887370","doi":"10.1002/sam.11161","title":"A survey on unsupervised outlier detection in high‐dimensional numerical data","year":2012,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":856,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Outlier; Curse of dimensionality; Anomaly detection; Computer science; Linear subspace; Data mining; Euclidean distance; Clustering high-dimensional data; Task (project management); Machine learning; Artificial intelligence; Mathematics; Cluster analysis","score_opus":0.09488812902028622,"score_gpt":0.35905860971230336,"score_spread":0.26417048069201715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2015887370","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008357667,0.031888574,0.9557455,0.00083938637,0.00023103126,0.00010553602,0.00021387447,0.0009865152,0.0016319298],"genre_scores_gemma":[0.14588901,0.079474404,0.7671575,0.000796696,0.0015108039,0.00037022942,0.0021607631,0.0004143598,0.002226245],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99252605,0.00208858,0.0009426077,0.0013323259,0.0028796082,0.00023078061],"domain_scores_gemma":[0.9831366,0.0097522335,0.0012652454,0.0018656678,0.0037637833,0.00021654562],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005236827,0.0014622658,0.0034026222,0.0063918596,0.00069477345,0.0031774682,0.0036508786,0.0021284372,0.0010046771],"category_scores_gemma":[0.023527477,0.00087462686,0.0021032458,0.011766427,0.0014846185,0.0040211147,0.0018364987,0.0018035871,0.0012258283],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013341391,0.00020417244,0.009923363,0.0030635404,0.00043332804,0.00028347835,0.00033958213,0.03874382,0.003552331,0.021697154,0.008114599,0.9135113],"study_design_scores_gemma":[0.000074002775,0.00050351594,0.014192147,0.0017910168,0.00038043514,0.0035120558,0.00075014256,0.69752085,0.022324441,0.11125819,0.14742187,0.00027134895],"about_ca_topic_score_codex":0.0010503958,"about_ca_topic_score_gemma":0.0007322573,"teacher_disagreement_score":0.0063918596,"about_ca_system_score_codex":0.0007522406,"about_ca_system_score_gemma":0.0014161491,"threshold_uncertainty_score":0.027695298},"labels":[],"label_agreement":null},{"id":"W2053843934","doi":"10.1002/sam.11149","title":"Nearest‐neighbors medians clustering","year":2012,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Bayesian Methods and Mixture Models","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia Hospital","funders":"","keywords":"Mathematics; Median; Uniqueness; Partition (number theory); Univariate; Cluster analysis; Nonparametric statistics; Fixed point; Sample (material); Algorithm; Combinatorics; Statistics; Mathematical analysis","score_opus":0.09250896968680458,"score_gpt":0.3798075243016192,"score_spread":0.28729855461481457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053843934","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014751137,0.00055232557,0.9820738,0.00021197843,0.00006909579,0.000084071566,0.0001642186,0.00044025743,0.0016531105],"genre_scores_gemma":[0.36948794,0.00045714184,0.6232173,0.00014087246,0.00024618502,0.00032261503,0.001232778,0.0002141434,0.0046811327],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961377,0.0013338872,0.00019480263,0.0009359698,0.0011754384,0.00022218682],"domain_scores_gemma":[0.9976672,0.00083948235,0.00028589106,0.00047914055,0.00062844675,0.00009988685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029983872,0.00076772505,0.0021796704,0.00320117,0.0015293215,0.0018924266,0.0033411232,0.0016860323,0.00271583],"category_scores_gemma":[0.008012608,0.00066538993,0.0014696948,0.0027663498,0.0011974991,0.0016241595,0.0018387563,0.001542725,0.0015172788],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005284623,0.00021093703,0.004884104,0.0001882128,0.00037624003,0.000087372995,0.0003961672,0.50107825,0.0040875208,0.055446304,0.008530873,0.42418557],"study_design_scores_gemma":[0.000035557903,0.000048888716,0.0011465843,0.000028432545,0.000025245177,0.00005409336,0.00006779867,0.94609916,0.0016485482,0.045188535,0.0056144847,0.000042735704],"about_ca_topic_score_codex":0.006140696,"about_ca_topic_score_gemma":0.0054537966,"teacher_disagreement_score":0.006140696,"about_ca_system_score_codex":0.0014280084,"about_ca_system_score_gemma":0.001398998,"threshold_uncertainty_score":0.01585716},"labels":[],"label_agreement":null},{"id":"W2153591306","doi":"10.1002/sam.11270","title":"Principal axes analysis of symbolic histogram variables","year":2015,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Sensory Analysis and Statistical Methods","field":"Agricultural and Biological Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Histogram; Mathematics; Principal component analysis; Estimator; Symbolic data analysis; Quantile; Pattern recognition (psychology); Histogram matching; Statistics; Artificial intelligence; Computer science; Image (mathematics)","score_opus":0.2031330746131957,"score_gpt":0.39143483193317025,"score_spread":0.18830175731997453,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2153591306","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072438405,0.00019684959,0.9892021,0.00008845333,0.000112960595,0.000099544304,0.00048320362,0.0008233862,0.0017496622],"genre_scores_gemma":[0.13582605,0.00044547606,0.8573837,0.00007092804,0.0002192498,0.00055227155,0.0014762465,0.0004568007,0.0035692456],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99795234,0.0004353745,0.00015494943,0.00042238005,0.000875216,0.00015969906],"domain_scores_gemma":[0.9973137,0.0007273058,0.00032738407,0.0004125481,0.0011081937,0.00011083929],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012771301,0.001224319,0.0008704839,0.003786103,0.00079329434,0.0020973154,0.00094796624,0.00046692503,0.009942176],"category_scores_gemma":[0.00748667,0.00033503887,0.0012453016,0.0046083275,0.0011566184,0.0019863616,0.0012957009,0.001615542,0.0027767627],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026194303,0.00014837316,0.007281527,0.00047050367,0.0001883117,0.00022024392,0.00048322033,0.021493027,0.021271473,0.051533516,0.008064814,0.88858306],"study_design_scores_gemma":[0.00011255096,0.000483519,0.030017627,0.00019547787,0.00021621442,0.0006941239,0.0008618368,0.70974797,0.035029035,0.14296176,0.07933755,0.00034230435],"about_ca_topic_score_codex":0.0022334484,"about_ca_topic_score_gemma":0.0018238244,"teacher_disagreement_score":0.009942176,"about_ca_system_score_codex":0.00046844347,"about_ca_system_score_gemma":0.0014238951,"threshold_uncertainty_score":0.03325987},"labels":[],"label_agreement":null},{"id":"W2407525010","doi":"10.1002/sam.11313","title":"Fast robust SUR with economical and actuarial applications","year":2016,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Advanced Statistical Methods and Models","field":"Mathematics","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Outlier; Estimator; Computer science; Robust statistics; Covariance; Multivariate statistics; Robust regression; Generalization; Algorithm; Econometrics; Mathematics; Statistics; Artificial intelligence","score_opus":0.224123729836376,"score_gpt":0.43411663687616964,"score_spread":0.20999290703979365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407525010","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010344609,0.0003532468,0.9864064,0.00023563928,0.00004374505,0.000042392043,0.00016010678,0.0010967456,0.0013171186],"genre_scores_gemma":[0.37391344,0.0004126872,0.6202222,0.00021663863,0.00015811951,0.00021807493,0.0008833096,0.00041580634,0.0035597836],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99752563,0.0010584525,0.00015983703,0.00041624965,0.00068512384,0.00015475502],"domain_scores_gemma":[0.9858241,0.009470685,0.0010022793,0.0021718803,0.0013326276,0.00019828891],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005424934,0.00090143346,0.0013878524,0.001950498,0.0006524259,0.0017047825,0.0018839433,0.0015574248,0.0059403623],"category_scores_gemma":[0.02757612,0.0008254966,0.0011407514,0.0016377394,0.0010988414,0.0025406154,0.0037102252,0.0019237073,0.0012280021],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012724598,0.000056345747,0.002977173,0.000110461944,0.000083438645,0.00014534828,0.00007187643,0.7647261,0.00097192114,0.056514684,0.003019692,0.17119575],"study_design_scores_gemma":[0.0000063119487,0.000018912664,0.00024349752,0.000007671613,0.0000037662626,0.000048471396,0.000010236085,0.9788608,0.0004301993,0.019226067,0.0011361296,0.000007985567],"about_ca_topic_score_codex":0.005784015,"about_ca_topic_score_gemma":0.0053958106,"teacher_disagreement_score":0.0059403623,"about_ca_system_score_codex":0.0011453971,"about_ca_system_score_gemma":0.0020548813,"threshold_uncertainty_score":0.0286901},"labels":[],"label_agreement":null},{"id":"W2509008132","doi":"10.1002/sam.11320","title":"Biased penalty calls in the National Hockey League","year":2016,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Sports Analytics and Performance","field":"Economics, Econometrics and Finance","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University; Université Laval","funders":"","keywords":"League; Situational ethics; Computer science; Penalty method; Data science; Psychology; Operations research; Applied psychology; Engineering; Mathematics; Social psychology","score_opus":0.1957237084336176,"score_gpt":0.35291672015557535,"score_spread":0.15719301172195774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2509008132","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9971296,0.00014089605,0.0002600214,0.0001971513,0.000018472676,0.0000061652827,0.00010177904,0.0000046139476,0.0021412603],"genre_scores_gemma":[0.9987318,0.000033031836,0.00006398091,0.00005810491,0.000017854465,0.000003964329,0.00010351427,0.0000031394507,0.0009845675],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9980459,0.00078053644,0.00007467649,0.00030190524,0.00042910085,0.0003677771],"domain_scores_gemma":[0.9863941,0.0046896287,0.00547496,0.00061746914,0.0009698921,0.001853938],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021823333,0.0001298217,0.00033909822,0.0007300033,0.0005931759,0.0012388333,0.00043340214,0.0006770232,0.004673342],"category_scores_gemma":[0.018252918,0.00011965025,0.0001709779,0.0005943284,0.00049436354,0.0004746556,0.0010980252,0.00084793166,0.0007396899],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000780197,0.0002944692,0.97612804,0.000019551026,0.00012240047,0.00025135023,0.0013233069,0.0016684401,0.000643333,0.0012550517,0.0019787224,0.01553514],"study_design_scores_gemma":[0.000017195414,0.00009108187,0.9953695,0.000010659334,0.000022623906,0.00006977312,0.001431241,0.0014630416,0.00009655872,0.00064085325,0.0007714995,0.000016026326],"about_ca_topic_score_codex":0.02118976,"about_ca_topic_score_gemma":0.047903977,"teacher_disagreement_score":0.02118976,"about_ca_system_score_codex":0.0009906297,"about_ca_system_score_gemma":0.0005735671,"threshold_uncertainty_score":0.042132795},"labels":[],"label_agreement":null},{"id":"W2790407909","doi":"10.1002/sam.11373","title":"Building cancer prognosis systems with survival function clusters","year":2018,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Bayesian Methods and Mixture Models","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Homogeneous; Cluster analysis; Computer science; Cancer; Lung cancer; Cluster (spacecraft); Population; Medicine; Covariate; Demographics; Survival analysis; Data mining; Oncology; Internal medicine; Artificial intelligence; Mathematics; Machine learning; Demography","score_opus":0.08062547735272997,"score_gpt":0.3739859447699232,"score_spread":0.29336046741719324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2790407909","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05409454,0.0003149386,0.93589664,0.0006198825,0.000049700488,0.00034273573,0.0025631522,0.004890005,0.0012283691],"genre_scores_gemma":[0.39025274,0.00028498913,0.600613,0.00018056395,0.00005897806,0.000619005,0.0065162317,0.00022472456,0.0012497408],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99846566,0.0004334353,0.00015206264,0.00057015097,0.0002763759,0.00010229625],"domain_scores_gemma":[0.9959746,0.0021760338,0.00045190597,0.00046032752,0.00082058145,0.00011663537],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002925085,0.0010360936,0.0010672662,0.0028260117,0.0009778735,0.0017510342,0.0013868633,0.0010772819,0.0021604358],"category_scores_gemma":[0.011584404,0.00089044514,0.0018620383,0.0021447334,0.0005433974,0.0018571386,0.0017960005,0.0012702462,0.0010437672],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029601087,0.00018282604,0.025336437,0.00019698126,0.00037505204,0.00018820478,0.0005229429,0.7562686,0.00151226,0.014389815,0.0067838975,0.19394688],"study_design_scores_gemma":[0.000012630329,0.000027982675,0.0013527028,0.00002016799,0.000037767626,0.000029927152,0.000057989968,0.9808528,0.0006288691,0.01539172,0.0015721129,0.000015418935],"about_ca_topic_score_codex":0.020749634,"about_ca_topic_score_gemma":0.018760303,"teacher_disagreement_score":0.020749634,"about_ca_system_score_codex":0.0019193854,"about_ca_system_score_gemma":0.0020527677,"threshold_uncertainty_score":0.04125768},"labels":[],"label_agreement":null},{"id":"W2793406240","doi":"10.1002/sam.11371","title":"Informative priors in Bayesian inference and computation","year":2018,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Gaussian Processes and Bayesian Inference","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Prior probability; Inference; Computer science; Bayesian inference; Prior information; Bayesian probability; Machine learning; Artificial intelligence; Computation; Algorithm","score_opus":0.052607974915139445,"score_gpt":0.37107473242317524,"score_spread":0.31846675750803577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2793406240","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00058936724,0.0058731646,0.98507684,0.0029737493,0.0002436986,0.00004811018,0.000091827824,0.00015424326,0.004948938],"genre_scores_gemma":[0.10726403,0.019182887,0.8630629,0.0022907455,0.002545733,0.00079524383,0.0003904993,0.00040872686,0.004059244],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97195226,0.020557739,0.0010657564,0.001979607,0.0040248926,0.00041966652],"domain_scores_gemma":[0.8945181,0.09573204,0.0020364278,0.0044092997,0.0027222105,0.0005820078],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.030178538,0.0024089073,0.003530806,0.0046468386,0.001670378,0.0072323224,0.004007127,0.0057372027,0.0063550826],"category_scores_gemma":[0.13269508,0.0020593188,0.0020964742,0.007163701,0.012050219,0.010978422,0.004980169,0.010570833,0.0023269774],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027498794,0.000025406533,0.0003402973,0.0003654783,0.00010460938,0.00008225329,0.00017595764,0.03371439,0.000094460745,0.91101176,0.0046453346,0.049412597],"study_design_scores_gemma":[0.000009739674,0.000006201929,0.00005173326,0.000088043555,0.000012565524,0.000024764575,0.000017489749,0.028885065,0.00007887816,0.9664587,0.0043512126,0.000015617006],"about_ca_topic_score_codex":0.0057364115,"about_ca_topic_score_gemma":0.00410517,"teacher_disagreement_score":0.030178538,"about_ca_system_score_codex":0.0042136326,"about_ca_system_score_gemma":0.0042519574,"threshold_uncertainty_score":0.15960121},"labels":[],"label_agreement":null},{"id":"W2953239640","doi":"10.1002/sam.11429","title":"Interactive volumetric segmentation for textile micro‐tomography data using wavelets and nonlocal means","year":2019,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Medical Image Segmentation Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Light Source (Canada)","funders":"Office of Science; National Aeronautics and Space Administration; U.S. Department of Energy","keywords":"Segmentation; Artificial intelligence; Voxel; Pattern recognition (psychology); Computer science; Wavelet; Discriminative model","score_opus":0.06895868629883593,"score_gpt":0.3872516588172325,"score_spread":0.31829297251839656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953239640","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03448027,0.00008948585,0.964518,0.000087240114,0.000008869659,0.000032676595,0.000060422713,0.00048457045,0.00023833306],"genre_scores_gemma":[0.29263753,0.00022115697,0.7054481,0.000068549336,0.00003850939,0.00011420908,0.00041516614,0.00021872428,0.00083802507],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994624,0.00013509022,0.00004178268,0.00012682166,0.00017975022,0.00005408091],"domain_scores_gemma":[0.9985846,0.0007777269,0.00019413249,0.00021815623,0.00017977605,0.000045607914],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011428725,0.00047498874,0.0006563159,0.0014995622,0.00027827176,0.00090148154,0.00070392614,0.0007198186,0.000820306],"category_scores_gemma":[0.0034249134,0.0003644083,0.0008205938,0.00093461387,0.00051915814,0.000870676,0.00082512345,0.000731442,0.00037567562],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004622218,0.00016390311,0.00576862,0.00028660588,0.00013199949,0.00021827857,0.000419146,0.1804946,0.23901372,0.006018571,0.001703135,0.5653192],"study_design_scores_gemma":[0.000007712583,0.00004485372,0.0021116978,0.000010477745,0.000012848553,0.00014331155,0.000059163787,0.95642066,0.037053276,0.0031630488,0.00095807924,0.000014779681],"about_ca_topic_score_codex":0.0010706175,"about_ca_topic_score_gemma":0.0027257926,"teacher_disagreement_score":0.0014995622,"about_ca_system_score_codex":0.00049895566,"about_ca_system_score_gemma":0.0005287782,"threshold_uncertainty_score":0.0060441494},"labels":[],"label_agreement":null},{"id":"W2964276935","doi":"10.1002/sam.11410","title":"Pruning variable selection ensembles","year":2019,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Face and Expression Recognition","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"China Scholarship Council; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Computer science; Benchmark (surveying); Sorting; Selection (genetic algorithm); Ensemble learning; Lasso (programming language); Context (archaeology); Stability (learning theory); Pruning; Artificial intelligence; Feature selection; Machine learning; Process (computing); Boosting (machine learning); Variable (mathematics); Algorithm; Mathematics","score_opus":0.04718228723840193,"score_gpt":0.3264302161271736,"score_spread":0.27924792888877165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964276935","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03536737,0.0011437994,0.9601498,0.0001497635,0.0001178747,0.00009099136,0.00014646791,0.0005589525,0.0022749596],"genre_scores_gemma":[0.58169585,0.0012718781,0.409479,0.0004239248,0.0003097805,0.00050484913,0.001787316,0.00022190591,0.0043054377],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973104,0.00091238116,0.00015339801,0.0004540631,0.0009594172,0.00021037045],"domain_scores_gemma":[0.99625623,0.0019990422,0.00024486484,0.0005240164,0.0008645892,0.00011125089],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003497677,0.0013108611,0.0021394722,0.0019017,0.0008811745,0.0010240859,0.001485155,0.0009149514,0.0015329915],"category_scores_gemma":[0.009722313,0.00042544148,0.0010487525,0.0016236394,0.0005038775,0.0011963482,0.0020783322,0.0011523498,0.0006498214],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023899195,0.00013203299,0.008527879,0.00021163239,0.00039220642,0.00032132506,0.00019287805,0.39397752,0.011165014,0.015246093,0.007589722,0.5620047],"study_design_scores_gemma":[0.000017383682,0.00009861111,0.0011253643,0.000035314857,0.00007859714,0.00013869895,0.00003924466,0.9809245,0.003909572,0.009699439,0.003918064,0.000015364838],"about_ca_topic_score_codex":0.0014269366,"about_ca_topic_score_gemma":0.0024504412,"teacher_disagreement_score":0.003497677,"about_ca_system_score_codex":0.00039521218,"about_ca_system_score_gemma":0.0009934779,"threshold_uncertainty_score":0.018497705},"labels":[],"label_agreement":null},{"id":"W2998595054","doi":"10.1002/sam.11445","title":"A new method for performance analysis in nonlinear dimensionality reduction","year":2020,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Face and Expression Recognition","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Dimensionality reduction; Skewness; Benchmark (surveying); Computer science; Curse of dimensionality; Intrinsic dimension; Measure (data warehouse); Dimension (graph theory); Rank (graph theory); Pattern recognition (psychology); Data mining; Artificial intelligence; Algorithm; Mathematics; Statistics; Geography","score_opus":0.10891862325224824,"score_gpt":0.3994450036008474,"score_spread":0.2905263803485991,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2998595054","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002649851,0.00034016772,0.994759,0.00011009024,0.00007879936,0.00006681177,0.00010190328,0.0006176147,0.0012756558],"genre_scores_gemma":[0.21857508,0.00063873,0.77299064,0.00022918356,0.0006877313,0.00091881474,0.0007089569,0.00087570876,0.0043751462],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9890682,0.004390453,0.0005110121,0.0012453237,0.0044297096,0.0003552926],"domain_scores_gemma":[0.9817077,0.010372656,0.0015737473,0.003036861,0.0030740977,0.00023485256],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01021679,0.002168181,0.0017878796,0.005705691,0.00074387615,0.002525827,0.0015791865,0.0011892999,0.0042244326],"category_scores_gemma":[0.035451893,0.00045664335,0.0013879886,0.003139299,0.0020761024,0.0030272824,0.0020573097,0.003058833,0.0015861063],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028954333,0.0002487956,0.0034936802,0.00062862824,0.00041860796,0.00016717466,0.0003268704,0.30477327,0.020841176,0.18875948,0.01309599,0.46695673],"study_design_scores_gemma":[0.000015394597,0.00012306105,0.0011211665,0.000034870653,0.000032952397,0.000102547245,0.000025444559,0.9640777,0.004762654,0.024428776,0.0052177566,0.000057612182],"about_ca_topic_score_codex":0.0019242092,"about_ca_topic_score_gemma":0.00092852773,"teacher_disagreement_score":0.01021679,"about_ca_system_score_codex":0.0016067581,"about_ca_system_score_gemma":0.0014180243,"threshold_uncertainty_score":0.054032207},"labels":[],"label_agreement":null},{"id":"W3169835580","doi":"10.1002/sam.11530","title":"Penalized composite likelihood for colored graphical Gaussian models","year":2021,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lunenfeld-Tanenbaum Research Institute; Mount Sinai Hospital; York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Graphical model; Computer science; Gaussian; Algorithm; Matrix (chemical analysis); Estimator; Selection (genetic algorithm); Model selection; Colored; Pattern recognition (psychology); Data mining; Mathematical optimization; Artificial intelligence; Mathematics; Statistics","score_opus":0.06367776533288279,"score_gpt":0.3649556518565307,"score_spread":0.30127788652364795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3169835580","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018940548,0.000056915822,0.9975902,0.00006969097,0.0000094559655,0.00002135066,0.00003698567,0.00012047723,0.00020092401],"genre_scores_gemma":[0.2874942,0.00033011995,0.7074943,0.0002977165,0.00014398785,0.00054798776,0.00090500026,0.00031637598,0.00247031],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9894287,0.0074751466,0.00028254013,0.001047378,0.0014130821,0.0003533167],"domain_scores_gemma":[0.96527106,0.028217643,0.001925243,0.0017421724,0.0023124528,0.00053142407],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014401883,0.0012174492,0.0020974353,0.002598391,0.0007950989,0.0024596294,0.0035474622,0.0018290961,0.003768282],"category_scores_gemma":[0.04660239,0.0010549026,0.0017898388,0.0021456012,0.0022157463,0.002195837,0.0028988735,0.003098829,0.0008342588],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025949345,0.00008606183,0.0021224348,0.00020622335,0.00020502009,0.00017601141,0.000121365025,0.7847523,0.0015751957,0.1359517,0.0027937125,0.07175048],"study_design_scores_gemma":[0.000012606899,0.000016217866,0.00012991199,0.000010467825,0.0000072954363,0.000017710243,0.0000050498365,0.97696257,0.00025642433,0.022238148,0.00033304995,0.000010567749],"about_ca_topic_score_codex":0.0031077953,"about_ca_topic_score_gemma":0.0025495915,"teacher_disagreement_score":0.014401883,"about_ca_system_score_codex":0.0017383785,"about_ca_system_score_gemma":0.0026028585,"threshold_uncertainty_score":0.07616532},"labels":[],"label_agreement":null},{"id":"W3176140770","doi":"10.1002/sam.11643","title":"Stratified learning: A general‐purpose statistical method for improved learning under covariate shift","year":2023,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Statistical Methods and Inference","field":"Mathematics","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Science and Technology Facilities Council; Engineering and Physical Sciences Research Council; Natural Sciences and Engineering Research Council of Canada","keywords":"Covariate; Computer science; Inference; Weighting; Causal inference; Machine learning; Artificial intelligence; Set (abstract data type); Statistics; Mathematics","score_opus":0.2786726334383422,"score_gpt":0.4905551637432045,"score_spread":0.21188253030486232,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176140770","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034577833,0.00012279225,0.99531955,0.00016734994,0.000030002071,0.0000561468,0.00006989782,0.0005694278,0.00020697463],"genre_scores_gemma":[0.20139425,0.00031560092,0.79322743,0.0007017742,0.00029631666,0.0006037266,0.00082489196,0.00042324155,0.0022127556],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99359363,0.0040614414,0.00025051553,0.0008848854,0.0009782773,0.00023127133],"domain_scores_gemma":[0.98322463,0.009891604,0.00091370253,0.0035332441,0.0019652012,0.00047166203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014644864,0.001297411,0.0019431084,0.0017785254,0.00091263064,0.0014016471,0.003492263,0.0018157121,0.0039002188],"category_scores_gemma":[0.0373327,0.0008851835,0.00183584,0.0016607763,0.001986333,0.0024925652,0.0037753275,0.0033833492,0.0015088272],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007765164,0.00045311492,0.012899494,0.00043129886,0.00061706634,0.00024173863,0.0005038273,0.31870133,0.0073488746,0.10181591,0.016655318,0.5395555],"study_design_scores_gemma":[0.0000887441,0.00013072157,0.00068383495,0.000034630804,0.000046988735,0.000052023745,0.000016647447,0.9315642,0.0018165726,0.06271625,0.0028264017,0.00002304657],"about_ca_topic_score_codex":0.0026672105,"about_ca_topic_score_gemma":0.0033635965,"teacher_disagreement_score":0.014644864,"about_ca_system_score_codex":0.0010421417,"about_ca_system_score_gemma":0.003211251,"threshold_uncertainty_score":0.077450335},"labels":[],"label_agreement":null},{"id":"W3195567601","doi":"10.1002/sam.11543","title":"Parallel coordinate order for<scp>high‐dimensional</scp>data","year":2021,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Sensory Analysis and Statistical Methods","field":"Agricultural and Biological Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Parallel coordinates; Computer science; Visualization; Coordinate descent; Data mining; Dimension (graph theory); Data visualization; Data structure; Theoretical computer science; Algorithm; Mathematics","score_opus":0.13471235951114346,"score_gpt":0.3734952579050034,"score_spread":0.2387828983938599,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3195567601","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020303715,0.0001645806,0.975043,0.00046097528,0.000045842156,0.00013863308,0.0006891608,0.0012956767,0.0018583861],"genre_scores_gemma":[0.16667588,0.00027407138,0.82911915,0.000090853755,0.00006078669,0.00027869048,0.0013557719,0.00040696596,0.0017378277],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983063,0.0006229086,0.00014058113,0.00026013938,0.00055324385,0.00011686038],"domain_scores_gemma":[0.9963748,0.0013779459,0.0005104963,0.0007085486,0.0008554025,0.000172822],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017078967,0.0009527612,0.0009561814,0.0023572105,0.00088458526,0.002576103,0.00083881064,0.0006479461,0.0046051242],"category_scores_gemma":[0.006837036,0.0003470752,0.0008801071,0.003989013,0.0013099012,0.0022442662,0.0015067274,0.0015685814,0.0009851087],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005588702,0.00030103995,0.0073921154,0.0009248366,0.00011578022,0.0007745127,0.00092965807,0.24271841,0.031753786,0.2654188,0.027377605,0.42173454],"study_design_scores_gemma":[0.000051493134,0.00018203903,0.002510826,0.00006360895,0.000024559151,0.00037198592,0.00032168254,0.83915323,0.014541459,0.120902635,0.021799909,0.000076612756],"about_ca_topic_score_codex":0.004329857,"about_ca_topic_score_gemma":0.0055644023,"teacher_disagreement_score":0.0046051242,"about_ca_system_score_codex":0.001183424,"about_ca_system_score_gemma":0.0020537993,"threshold_uncertainty_score":0.015405655},"labels":[],"label_agreement":null},{"id":"W3208584122","doi":"10.1002/sam.11555","title":"A family of mixture models for biclustering","year":2021,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Bayesian Methods and Mixture Models","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Biclustering; Cluster analysis; Diagonal; Covariance matrix; Latent variable; Covariance; Block matrix; Mathematics; Algorithm; Computer science; Matrix (chemical analysis); Pattern recognition (psychology); Data mining; Artificial intelligence; Statistics; Correlation clustering; CURE data clustering algorithm; Eigenvalues and eigenvectors","score_opus":0.11421512759505388,"score_gpt":0.3767657191606953,"score_spread":0.2625505915656414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3208584122","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001144631,0.0003454273,0.99744,0.00018271117,0.00004647461,0.00007805421,0.00014422937,0.00028799626,0.00033045522],"genre_scores_gemma":[0.09541809,0.0015049439,0.8919931,0.00073084666,0.00025703598,0.0020408381,0.0023854529,0.00072272384,0.004946906],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9855181,0.009614489,0.0006718133,0.0019813336,0.001772003,0.00044230127],"domain_scores_gemma":[0.9767754,0.016494531,0.0013665542,0.002131968,0.0027916466,0.00043984217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019593708,0.0025011054,0.004334419,0.004719516,0.002376675,0.003423415,0.006621443,0.0030451412,0.0062501645],"category_scores_gemma":[0.045930173,0.0022530635,0.005704847,0.0051189116,0.0028047988,0.003717037,0.004835012,0.0057336516,0.0035615258],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030966176,0.00015793035,0.0044538784,0.0006734635,0.000979022,0.00029727258,0.0008306585,0.5296517,0.0016400848,0.31741944,0.0129406,0.13064627],"study_design_scores_gemma":[0.000017952088,0.000027334121,0.00024303113,0.00006908569,0.000040905532,0.00007108835,0.00002894756,0.90948546,0.00022571327,0.08667878,0.0030789243,0.000032674747],"about_ca_topic_score_codex":0.007340168,"about_ca_topic_score_gemma":0.007497487,"teacher_disagreement_score":0.019593708,"about_ca_system_score_codex":0.0021054146,"about_ca_system_score_gemma":0.0026262077,"threshold_uncertainty_score":0.103622615},"labels":[],"label_agreement":null},{"id":"W4206662956","doi":"10.1002/sam.11466","title":"Issue Information","year":2021,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Information retrieval","score_opus":0.07724045282164911,"score_gpt":0.3726486626682964,"score_spread":0.2954082098466473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4206662956","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0002752305,0.0014051053,0.0016983212,0.0042212596,0.017641399,0.00058971596,0.023178548,0.0034422574,0.9475481],"genre_scores_gemma":[0.000533255,0.00079078675,0.00064055773,0.0015406887,0.002064275,0.00015564225,0.0088901045,0.0007366248,0.9846482],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998796,0.0001342282,0.000093693154,0.00017284113,0.0006708102,0.00013240805],"domain_scores_gemma":[0.99544215,0.000570911,0.0002024367,0.00047142492,0.0020362218,0.0012769281],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001392038,0.0017588414,0.0016988203,0.0060343603,0.0013557322,0.008070702,0.0021266777,0.00265628,0.92552346],"category_scores_gemma":[0.0079164505,0.0006901106,0.0011467219,0.004686335,0.0005294295,0.0033916363,0.0027040371,0.0019214455,0.9129701],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016792334,0.000027479466,0.000042553966,0.00016781472,0.0000021315304,0.000016025831,0.000005779239,0.00003091574,0.00011656424,0.0008211872,0.9528662,0.045886487],"study_design_scores_gemma":[0.00001779712,0.000017119663,0.00019907406,0.00010224748,0.000002557078,0.000023525039,0.000017181621,0.000054492113,0.000071345,0.0011801494,0.99830854,0.0000059386525],"about_ca_topic_score_codex":0.0014893331,"about_ca_topic_score_gemma":0.0032755472,"teacher_disagreement_score":0.07447654,"about_ca_system_score_codex":0.000983167,"about_ca_system_score_gemma":0.0028366582,"threshold_uncertainty_score":0.10623175},"labels":[],"label_agreement":null},{"id":"W4210695710","doi":"10.1002/sam.11427","title":"Issue Information","year":2020,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; Data science; World Wide Web","score_opus":0.08550868619541674,"score_gpt":0.3686068909417464,"score_spread":0.2830982047463297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210695710","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00027672268,0.0013541734,0.001703175,0.004088131,0.017366836,0.00056218094,0.02298662,0.003342444,0.9483197],"genre_scores_gemma":[0.0005282513,0.00075410557,0.00063574285,0.0014167732,0.0018316957,0.00014308663,0.008770358,0.00067396916,0.985246],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988017,0.00013230708,0.000090055895,0.0001683756,0.000675938,0.00013161366],"domain_scores_gemma":[0.99534035,0.00055490446,0.00019841177,0.0004624762,0.0021202012,0.0013236412],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0014310633,0.001664715,0.0015620437,0.0058080773,0.0013100061,0.007997186,0.0020408118,0.002592302,0.92210764],"category_scores_gemma":[0.008057451,0.0006340004,0.0011086134,0.0043734265,0.0005280851,0.0032663841,0.0026291008,0.0018657091,0.90687567],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017473316,0.000026307489,0.000042200925,0.00016494922,0.0000020089033,0.000015299198,0.000005759341,0.00003228389,0.00011793516,0.00089069695,0.95210177,0.046583228],"study_design_scores_gemma":[0.000017054528,0.000016867365,0.00019071034,0.000097911216,0.0000023697078,0.000021866183,0.000016520406,0.00005500304,0.000069959446,0.0012025037,0.99830365,0.0000055239257],"about_ca_topic_score_codex":0.001588501,"about_ca_topic_score_gemma":0.0034349945,"teacher_disagreement_score":0.92210764,"about_ca_system_score_codex":0.0010148198,"about_ca_system_score_gemma":0.002950895,"threshold_uncertainty_score":0.11110401},"labels":[],"label_agreement":null},{"id":"W4213289775","doi":"10.1002/sam.11425","title":"Issue Information","year":2020,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; World Wide Web; Data science","score_opus":0.08550868619541674,"score_gpt":0.3686068909417464,"score_spread":0.2830982047463297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4213289775","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00027672268,0.0013541734,0.001703175,0.004088131,0.017366836,0.00056218094,0.02298662,0.003342444,0.9483197],"genre_scores_gemma":[0.0005282513,0.00075410557,0.00063574285,0.0014167732,0.0018316957,0.00014308663,0.008770358,0.00067396916,0.985246],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988017,0.00013230708,0.000090055895,0.0001683756,0.000675938,0.00013161366],"domain_scores_gemma":[0.99534035,0.00055490446,0.00019841177,0.0004624762,0.0021202012,0.0013236412],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0014310633,0.001664715,0.0015620437,0.0058080773,0.0013100061,0.007997186,0.0020408118,0.002592302,0.92210764],"category_scores_gemma":[0.008057451,0.0006340004,0.0011086134,0.0043734265,0.0005280851,0.0032663841,0.0026291008,0.0018657091,0.90687567],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017473316,0.000026307489,0.000042200925,0.00016494922,0.0000020089033,0.000015299198,0.000005759341,0.00003228389,0.00011793516,0.00089069695,0.95210177,0.046583228],"study_design_scores_gemma":[0.000017054528,0.000016867365,0.00019071034,0.000097911216,0.0000023697078,0.000021866183,0.000016520406,0.00005500304,0.000069959446,0.0012025037,0.99830365,0.0000055239257],"about_ca_topic_score_codex":0.001588501,"about_ca_topic_score_gemma":0.0034349945,"teacher_disagreement_score":0.07789236,"about_ca_system_score_codex":0.0010148198,"about_ca_system_score_gemma":0.002950895,"threshold_uncertainty_score":0.11110401},"labels":[],"label_agreement":null},{"id":"W4214902008","doi":"10.1002/sam.11469","title":"Issue Information","year":2021,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Citation; Information retrieval; Data science; World Wide Web","score_opus":0.07724045282164911,"score_gpt":0.3726486626682964,"score_spread":0.2954082098466473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214902008","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0002752305,0.0014051053,0.0016983212,0.0042212596,0.017641399,0.00058971596,0.023178548,0.0034422574,0.9475481],"genre_scores_gemma":[0.000533255,0.00079078675,0.00064055773,0.0015406887,0.002064275,0.00015564225,0.0088901045,0.0007366248,0.9846482],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998796,0.0001342282,0.000093693154,0.00017284113,0.0006708102,0.00013240805],"domain_scores_gemma":[0.99544215,0.000570911,0.0002024367,0.00047142492,0.0020362218,0.0012769281],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001392038,0.0017588414,0.0016988203,0.0060343603,0.0013557322,0.008070702,0.0021266777,0.00265628,0.92552346],"category_scores_gemma":[0.0079164505,0.0006901106,0.0011467219,0.004686335,0.0005294295,0.0033916363,0.0027040371,0.0019214455,0.9129701],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016792334,0.000027479466,0.000042553966,0.00016781472,0.0000021315304,0.000016025831,0.000005779239,0.00003091574,0.00011656424,0.0008211872,0.9528662,0.045886487],"study_design_scores_gemma":[0.00001779712,0.000017119663,0.00019907406,0.00010224748,0.000002557078,0.000023525039,0.000017181621,0.000054492113,0.000071345,0.0011801494,0.99830854,0.0000059386525],"about_ca_topic_score_codex":0.0014893331,"about_ca_topic_score_gemma":0.0032755472,"teacher_disagreement_score":0.07447654,"about_ca_system_score_codex":0.000983167,"about_ca_system_score_gemma":0.0028366582,"threshold_uncertainty_score":0.10623175},"labels":[],"label_agreement":null},{"id":"W4231631488","doi":"10.1002/sam.10048","title":"Application of model selection technique in chemogenomic data analysis","year":2009,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; York University","funders":"National Cancer Institute; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Data mining; Dependency (UML); Model selection; Bayesian network; Selection (genetic algorithm); Data set; Construct (python library); Set (abstract data type); Bayesian information criterion; Bayesian probability; Artificial intelligence; Machine learning","score_opus":0.04998375380142114,"score_gpt":0.37110303310567316,"score_spread":0.32111927930425205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231631488","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0132766,0.00019097973,0.98515946,0.00030523134,0.00002983471,0.00013701762,0.00018463441,0.0005529424,0.0001633622],"genre_scores_gemma":[0.4114432,0.00028311447,0.5850589,0.00022760419,0.00010538684,0.0009694022,0.0012224866,0.00016889357,0.00052101014],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98771393,0.010090526,0.00038206545,0.00076044624,0.0008701402,0.00018285168],"domain_scores_gemma":[0.9651266,0.031446744,0.00096773426,0.00096776744,0.0012231383,0.0002680766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016691543,0.0015500586,0.002084939,0.0039066174,0.0012272642,0.0014270799,0.0014833586,0.0010201305,0.0013311482],"category_scores_gemma":[0.03416061,0.00061159115,0.0025609427,0.0029658657,0.0009698569,0.00084789854,0.0012375696,0.002442319,0.00034414066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059516914,0.00035440223,0.021800185,0.000391453,0.0027741662,0.0009023994,0.00032384114,0.7293355,0.0054432442,0.023331186,0.0047220206,0.21002646],"study_design_scores_gemma":[0.0000322515,0.00006228442,0.00075658306,0.000011349622,0.00006550119,0.000057039695,0.00001666766,0.9870194,0.0007168549,0.010822065,0.00042509587,0.000014888153],"about_ca_topic_score_codex":0.004244085,"about_ca_topic_score_gemma":0.0039705196,"teacher_disagreement_score":0.016691543,"about_ca_system_score_codex":0.0010521916,"about_ca_system_score_gemma":0.0026051856,"threshold_uncertainty_score":0.08827442},"labels":[],"label_agreement":null},{"id":"W4231764799","doi":"10.1002/sam.11387","title":"Issue Information","year":2019,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; World Wide Web; Data science","score_opus":0.07162451266438453,"score_gpt":0.3683286561938265,"score_spread":0.29670414352944197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231764799","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00030509027,0.0015456381,0.0019364037,0.0046779257,0.019253427,0.00059452513,0.023859447,0.0034126483,0.9444149],"genre_scores_gemma":[0.00057088496,0.0008376102,0.0006725755,0.0014766619,0.0021320586,0.00014946512,0.008793686,0.00066366384,0.9847035],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987562,0.00013935953,0.00010072144,0.00017840716,0.0006917742,0.0001335674],"domain_scores_gemma":[0.9951003,0.0006331515,0.00023471199,0.0004898531,0.0022032743,0.0013387529],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0014260353,0.0016498428,0.001585639,0.0061060237,0.0012603584,0.007965645,0.002035216,0.0025359157,0.9169632],"category_scores_gemma":[0.00851714,0.000628905,0.0010783955,0.00473976,0.0005429835,0.0032411567,0.0026259716,0.0018785497,0.8996377],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017285678,0.000028495815,0.000046994515,0.00018646594,0.0000022650017,0.000017323619,0.000006424985,0.000031857697,0.0001303429,0.0009515048,0.94789433,0.05068661],"study_design_scores_gemma":[0.00001648076,0.000017365617,0.00020987034,0.000109261986,0.0000025955026,0.0000254837,0.000017980912,0.00005588763,0.00007514352,0.0013246743,0.9981395,0.000005741392],"about_ca_topic_score_codex":0.001365952,"about_ca_topic_score_gemma":0.003058327,"teacher_disagreement_score":0.08303678,"about_ca_system_score_codex":0.0010153034,"about_ca_system_score_gemma":0.00290036,"threshold_uncertainty_score":0.11844182},"labels":[],"label_agreement":null},{"id":"W4236215018","doi":"10.1002/sam.11360","title":"Issue Information","year":2018,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; World Wide Web; Data science","score_opus":0.07753281751408107,"score_gpt":0.3766135914705343,"score_spread":0.2990807739564532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4236215018","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00035678377,0.0016089336,0.0023318867,0.005258304,0.019779647,0.0006447342,0.02798154,0.0043958435,0.93764234],"genre_scores_gemma":[0.00077471766,0.0009125715,0.0008407251,0.0017277214,0.0023637346,0.00018639365,0.010682073,0.0009811241,0.9815309],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986859,0.00015436999,0.00010865419,0.00020029003,0.0007078019,0.00014302102],"domain_scores_gemma":[0.9944587,0.00071134046,0.0002537616,0.0005984537,0.0024110153,0.0015667147],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0015373783,0.0016453845,0.0016302003,0.0057391645,0.0012899289,0.008288074,0.0021268143,0.0024793397,0.9239934],"category_scores_gemma":[0.009646704,0.000604908,0.00107815,0.004407087,0.0005231752,0.0034624524,0.00289475,0.0019930406,0.90701264],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017992032,0.000030394362,0.00005089748,0.00018750291,0.0000025080897,0.000017448403,0.00000729007,0.000032815566,0.00013161048,0.00091526005,0.94679123,0.051814925],"study_design_scores_gemma":[0.000016704402,0.000016572683,0.0001909207,0.000103823564,0.0000026602363,0.000024661716,0.00001850466,0.000058549933,0.000078525154,0.0013984925,0.99808466,0.0000058354685],"about_ca_topic_score_codex":0.0011925946,"about_ca_topic_score_gemma":0.002669039,"teacher_disagreement_score":0.07600659,"about_ca_system_score_codex":0.00096302014,"about_ca_system_score_gemma":0.0029552295,"threshold_uncertainty_score":0.10841417},"labels":[],"label_agreement":null},{"id":"W4239375881","doi":"10.1002/sam.11423","title":"Issue Information","year":2020,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; World Wide Web; Data science","score_opus":0.08550868619541674,"score_gpt":0.3686068909417464,"score_spread":0.2830982047463297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4239375881","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00027672268,0.0013541734,0.001703175,0.004088131,0.017366836,0.00056218094,0.02298662,0.003342444,0.9483197],"genre_scores_gemma":[0.0005282513,0.00075410557,0.00063574285,0.0014167732,0.0018316957,0.00014308663,0.008770358,0.00067396916,0.985246],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988017,0.00013230708,0.000090055895,0.0001683756,0.000675938,0.00013161366],"domain_scores_gemma":[0.99534035,0.00055490446,0.00019841177,0.0004624762,0.0021202012,0.0013236412],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0014310633,0.001664715,0.0015620437,0.0058080773,0.0013100061,0.007997186,0.0020408118,0.002592302,0.92210764],"category_scores_gemma":[0.008057451,0.0006340004,0.0011086134,0.0043734265,0.0005280851,0.0032663841,0.0026291008,0.0018657091,0.90687567],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017473316,0.000026307489,0.000042200925,0.00016494922,0.0000020089033,0.000015299198,0.000005759341,0.00003228389,0.00011793516,0.00089069695,0.95210177,0.046583228],"study_design_scores_gemma":[0.000017054528,0.000016867365,0.00019071034,0.000097911216,0.0000023697078,0.000021866183,0.000016520406,0.00005500304,0.000069959446,0.0012025037,0.99830365,0.0000055239257],"about_ca_topic_score_codex":0.001588501,"about_ca_topic_score_gemma":0.0034349945,"teacher_disagreement_score":0.07789236,"about_ca_system_score_codex":0.0010148198,"about_ca_system_score_gemma":0.002950895,"threshold_uncertainty_score":0.11110401},"labels":[],"label_agreement":null},{"id":"W4240834168","doi":"10.1002/sam.10016","title":"Discovering and Exploiting Statistical Properties for Query Optimization in Relational Databases: A Survey","year":2009,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Query optimization; Relational database; Data mining; Relational database management system; Cardinality (data modeling); Relational model; Sargable; View; Information retrieval; Key (lock); Exploratory data analysis; Database; Database design; Web search query; Search engine","score_opus":0.15298434666212693,"score_gpt":0.3608533658037779,"score_spread":0.20786901914165098,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4240834168","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021059643,0.032690395,0.94013053,0.0016848986,0.0000504741,0.0001404281,0.00051780103,0.0015653785,0.0021604467],"genre_scores_gemma":[0.25008777,0.028280666,0.7178424,0.00036943302,0.00045623936,0.00026172126,0.0016980926,0.00042925682,0.0005744135],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9850631,0.0044922484,0.001897105,0.0020190512,0.0062407604,0.00028784914],"domain_scores_gemma":[0.942894,0.041401196,0.0038713892,0.0060590114,0.0053299167,0.0004446704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0151049765,0.0012439846,0.0028313515,0.007893068,0.00067638414,0.0047251075,0.0035229616,0.0011330151,0.0009160828],"category_scores_gemma":[0.038869172,0.0012212895,0.002275746,0.010706523,0.001963967,0.006773667,0.0019104696,0.0014397728,0.00060804695],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017733726,0.00033937476,0.030359227,0.0029945206,0.0005229906,0.00031016397,0.0006135264,0.060854036,0.0055500437,0.049316224,0.005040434,0.8439221],"study_design_scores_gemma":[0.00007955518,0.0003201259,0.0121947685,0.0005842703,0.00033046232,0.0014934334,0.0006876361,0.7886672,0.017060697,0.13855667,0.039809387,0.00021583392],"about_ca_topic_score_codex":0.0026868798,"about_ca_topic_score_gemma":0.0016583493,"teacher_disagreement_score":0.0151049765,"about_ca_system_score_codex":0.0012732174,"about_ca_system_score_gemma":0.0022995742,"threshold_uncertainty_score":0.079883695},"labels":[],"label_agreement":null},{"id":"W4244236454","doi":"10.1002/sam.11386","title":"Issue Information","year":2019,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Transportation Systems and Logistics","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; Data science; World Wide Web; Library science","score_opus":0.048545103820620786,"score_gpt":0.3262234257958047,"score_spread":0.27767832197518394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4244236454","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00029240976,0.0014007554,0.0015737825,0.0039727017,0.014446146,0.00047773178,0.026608463,0.0030216048,0.9482064],"genre_scores_gemma":[0.0006537021,0.0009407311,0.00060225365,0.0013596104,0.0018724067,0.00013872006,0.011623754,0.00067231857,0.9821364],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99873835,0.00014357531,0.00010996224,0.00019550018,0.00067643874,0.00013612781],"domain_scores_gemma":[0.9956632,0.00056730927,0.00023227635,0.0005177436,0.0018681854,0.0011511701],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0012831814,0.0019218543,0.0017069897,0.0069519645,0.0012973208,0.008522432,0.0020883677,0.0025323494,0.9188953],"category_scores_gemma":[0.007609891,0.0006615704,0.0010286273,0.0060897963,0.0005459241,0.0039018665,0.0028340425,0.0019247275,0.9087439],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001677475,0.000027255071,0.00005016131,0.00019684099,0.0000023385835,0.000017496826,0.0000075093612,0.000031189353,0.00013096178,0.0009717228,0.95067716,0.04787065],"study_design_scores_gemma":[0.000014498642,0.000016204045,0.00022135246,0.00010790543,0.000002768418,0.00002382029,0.000018747503,0.000053075823,0.000071778784,0.0011366714,0.9983278,0.0000053479116],"about_ca_topic_score_codex":0.0013627361,"about_ca_topic_score_gemma":0.002766717,"teacher_disagreement_score":0.081104696,"about_ca_system_score_codex":0.0009796027,"about_ca_system_score_gemma":0.0027812303,"threshold_uncertainty_score":0.115686},"labels":[],"label_agreement":null},{"id":"W4244973481","doi":"10.1002/sam.11358","title":"Issue Information","year":2018,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; Data science; World Wide Web","score_opus":0.07753281751408107,"score_gpt":0.3766135914705343,"score_spread":0.2990807739564532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4244973481","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00035678377,0.0016089336,0.0023318867,0.005258304,0.019779647,0.0006447342,0.02798154,0.0043958435,0.93764234],"genre_scores_gemma":[0.00077471766,0.0009125715,0.0008407251,0.0017277214,0.0023637346,0.00018639365,0.010682073,0.0009811241,0.9815309],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986859,0.00015436999,0.00010865419,0.00020029003,0.0007078019,0.00014302102],"domain_scores_gemma":[0.9944587,0.00071134046,0.0002537616,0.0005984537,0.0024110153,0.0015667147],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0015373783,0.0016453845,0.0016302003,0.0057391645,0.0012899289,0.008288074,0.0021268143,0.0024793397,0.9239934],"category_scores_gemma":[0.009646704,0.000604908,0.00107815,0.004407087,0.0005231752,0.0034624524,0.00289475,0.0019930406,0.90701264],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017992032,0.000030394362,0.00005089748,0.00018750291,0.0000025080897,0.000017448403,0.00000729007,0.000032815566,0.00013161048,0.00091526005,0.94679123,0.051814925],"study_design_scores_gemma":[0.000016704402,0.000016572683,0.0001909207,0.000103823564,0.0000026602363,0.000024661716,0.00001850466,0.000058549933,0.000078525154,0.0013984925,0.99808466,0.0000058354685],"about_ca_topic_score_codex":0.0011925946,"about_ca_topic_score_gemma":0.002669039,"teacher_disagreement_score":0.07600659,"about_ca_system_score_codex":0.00096302014,"about_ca_system_score_gemma":0.0029552295,"threshold_uncertainty_score":0.10841417},"labels":[],"label_agreement":null},{"id":"W4245052885","doi":"10.1002/sam.11384","title":"Issue Information","year":2019,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; World Wide Web; Data science","score_opus":0.07162451266438453,"score_gpt":0.3683286561938265,"score_spread":0.29670414352944197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4245052885","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00030509027,0.0015456381,0.0019364037,0.0046779257,0.019253427,0.00059452513,0.023859447,0.0034126483,0.9444149],"genre_scores_gemma":[0.00057088496,0.0008376102,0.0006725755,0.0014766619,0.0021320586,0.00014946512,0.008793686,0.00066366384,0.9847035],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987562,0.00013935953,0.00010072144,0.00017840716,0.0006917742,0.0001335674],"domain_scores_gemma":[0.9951003,0.0006331515,0.00023471199,0.0004898531,0.0022032743,0.0013387529],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0014260353,0.0016498428,0.001585639,0.0061060237,0.0012603584,0.007965645,0.002035216,0.0025359157,0.9169632],"category_scores_gemma":[0.00851714,0.000628905,0.0010783955,0.00473976,0.0005429835,0.0032411567,0.0026259716,0.0018785497,0.8996377],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017285678,0.000028495815,0.000046994515,0.00018646594,0.0000022650017,0.000017323619,0.000006424985,0.000031857697,0.0001303429,0.0009515048,0.94789433,0.05068661],"study_design_scores_gemma":[0.00001648076,0.000017365617,0.00020987034,0.000109261986,0.0000025955026,0.0000254837,0.000017980912,0.00005588763,0.00007514352,0.0013246743,0.9981395,0.000005741392],"about_ca_topic_score_codex":0.001365952,"about_ca_topic_score_gemma":0.003058327,"teacher_disagreement_score":0.08303678,"about_ca_system_score_codex":0.0010153034,"about_ca_system_score_gemma":0.00290036,"threshold_uncertainty_score":0.11844182},"labels":[],"label_agreement":null},{"id":"W4245325236","doi":"10.1002/sam.11362","title":"Issue Information","year":2018,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; Data science; World Wide Web","score_opus":0.07753281751408107,"score_gpt":0.3766135914705343,"score_spread":0.2990807739564532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4245325236","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00035678377,0.0016089336,0.0023318867,0.005258304,0.019779647,0.0006447342,0.02798154,0.0043958435,0.93764234],"genre_scores_gemma":[0.00077471766,0.0009125715,0.0008407251,0.0017277214,0.0023637346,0.00018639365,0.010682073,0.0009811241,0.9815309],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986859,0.00015436999,0.00010865419,0.00020029003,0.0007078019,0.00014302102],"domain_scores_gemma":[0.9944587,0.00071134046,0.0002537616,0.0005984537,0.0024110153,0.0015667147],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0015373783,0.0016453845,0.0016302003,0.0057391645,0.0012899289,0.008288074,0.0021268143,0.0024793397,0.9239934],"category_scores_gemma":[0.009646704,0.000604908,0.00107815,0.004407087,0.0005231752,0.0034624524,0.00289475,0.0019930406,0.90701264],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017992032,0.000030394362,0.00005089748,0.00018750291,0.0000025080897,0.000017448403,0.00000729007,0.000032815566,0.00013161048,0.00091526005,0.94679123,0.051814925],"study_design_scores_gemma":[0.000016704402,0.000016572683,0.0001909207,0.000103823564,0.0000026602363,0.000024661716,0.00001850466,0.000058549933,0.000078525154,0.0013984925,0.99808466,0.0000058354685],"about_ca_topic_score_codex":0.0011925946,"about_ca_topic_score_gemma":0.002669039,"teacher_disagreement_score":0.07600659,"about_ca_system_score_codex":0.00096302014,"about_ca_system_score_gemma":0.0029552295,"threshold_uncertainty_score":0.10841417},"labels":[],"label_agreement":null},{"id":"W4254812759","doi":"10.1002/sam.11467","title":"Issue Information","year":2021,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; Data science; World Wide Web","score_opus":0.07724045282164911,"score_gpt":0.3726486626682964,"score_spread":0.2954082098466473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254812759","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0002752305,0.0014051053,0.0016983212,0.0042212596,0.017641399,0.00058971596,0.023178548,0.0034422574,0.9475481],"genre_scores_gemma":[0.000533255,0.00079078675,0.00064055773,0.0015406887,0.002064275,0.00015564225,0.0088901045,0.0007366248,0.9846482],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998796,0.0001342282,0.000093693154,0.00017284113,0.0006708102,0.00013240805],"domain_scores_gemma":[0.99544215,0.000570911,0.0002024367,0.00047142492,0.0020362218,0.0012769281],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001392038,0.0017588414,0.0016988203,0.0060343603,0.0013557322,0.008070702,0.0021266777,0.00265628,0.92552346],"category_scores_gemma":[0.0079164505,0.0006901106,0.0011467219,0.004686335,0.0005294295,0.0033916363,0.0027040371,0.0019214455,0.9129701],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016792334,0.000027479466,0.000042553966,0.00016781472,0.0000021315304,0.000016025831,0.000005779239,0.00003091574,0.00011656424,0.0008211872,0.9528662,0.045886487],"study_design_scores_gemma":[0.00001779712,0.000017119663,0.00019907406,0.00010224748,0.000002557078,0.000023525039,0.000017181621,0.000054492113,0.000071345,0.0011801494,0.99830854,0.0000059386525],"about_ca_topic_score_codex":0.0014893331,"about_ca_topic_score_gemma":0.0032755472,"teacher_disagreement_score":0.07447654,"about_ca_system_score_codex":0.000983167,"about_ca_system_score_gemma":0.0028366582,"threshold_uncertainty_score":0.10623175},"labels":[],"label_agreement":null},{"id":"W4255653278","doi":"10.1002/sam.11385","title":"Issue Information","year":2019,"lang":"en","type":"paratext","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Human auditory perception and evaluation","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Computer science; Information retrieval; Citation; Data science; World Wide Web","score_opus":0.07162451266438453,"score_gpt":0.3683286561938265,"score_spread":0.29670414352944197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4255653278","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00030509027,0.0015456381,0.0019364037,0.0046779257,0.019253427,0.00059452513,0.023859447,0.0034126483,0.9444149],"genre_scores_gemma":[0.00057088496,0.0008376102,0.0006725755,0.0014766619,0.0021320586,0.00014946512,0.008793686,0.00066366384,0.9847035],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987562,0.00013935953,0.00010072144,0.00017840716,0.0006917742,0.0001335674],"domain_scores_gemma":[0.9951003,0.0006331515,0.00023471199,0.0004898531,0.0022032743,0.0013387529],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0014260353,0.0016498428,0.001585639,0.0061060237,0.0012603584,0.007965645,0.002035216,0.0025359157,0.9169632],"category_scores_gemma":[0.00851714,0.000628905,0.0010783955,0.00473976,0.0005429835,0.0032411567,0.0026259716,0.0018785497,0.8996377],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017285678,0.000028495815,0.000046994515,0.00018646594,0.0000022650017,0.000017323619,0.000006424985,0.000031857697,0.0001303429,0.0009515048,0.94789433,0.05068661],"study_design_scores_gemma":[0.00001648076,0.000017365617,0.00020987034,0.000109261986,0.0000025955026,0.0000254837,0.000017980912,0.00005588763,0.00007514352,0.0013246743,0.9981395,0.000005741392],"about_ca_topic_score_codex":0.001365952,"about_ca_topic_score_gemma":0.003058327,"teacher_disagreement_score":0.08303678,"about_ca_system_score_codex":0.0010153034,"about_ca_system_score_gemma":0.00290036,"threshold_uncertainty_score":0.11844182},"labels":[],"label_agreement":null},{"id":"W4392370559","doi":"10.1002/sam.11668","title":"Marginal clustered multistate models for longitudinal progressive processes with informative cluster size","year":2024,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Data-Driven Disease Surveillance","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Public Health Ontario; University of Toronto","funders":"","keywords":"Resampling; Inference; Cluster (spacecraft); Computer science; Statistics; Data mining; Process (computing); Econometrics; Artificial intelligence; Mathematics","score_opus":0.0727690246232361,"score_gpt":0.38694199091094134,"score_spread":0.31417296628770525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392370559","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062228292,0.000355017,0.9351175,0.00068505754,0.0000716671,0.00017669551,0.000461199,0.00022995361,0.00067469594],"genre_scores_gemma":[0.83237666,0.00047862402,0.15967183,0.0003428565,0.00013680145,0.00073583913,0.0011734591,0.00012391567,0.0049599702],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942151,0.003345849,0.0002310043,0.001401948,0.00044289214,0.0003632725],"domain_scores_gemma":[0.9487565,0.040121846,0.003570903,0.0035851768,0.0030809145,0.0008847097],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023020994,0.0010333379,0.0021872004,0.002085357,0.0009789724,0.0024998446,0.004688284,0.0020436468,0.004307344],"category_scores_gemma":[0.05482862,0.00093773665,0.0028157749,0.0019993351,0.00336074,0.003019294,0.0029391663,0.0036859787,0.00049658614],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030122916,0.0001332618,0.018704083,0.00019465349,0.00048030354,0.0002525447,0.00074638054,0.7340716,0.0006175365,0.2183296,0.002304114,0.023864713],"study_design_scores_gemma":[0.000021119515,0.00003006381,0.0014553306,0.000025141588,0.000043984746,0.000018834904,0.000056560002,0.9503825,0.00014383816,0.047328193,0.00047051912,0.000023860353],"about_ca_topic_score_codex":0.013356085,"about_ca_topic_score_gemma":0.011313359,"teacher_disagreement_score":0.023020994,"about_ca_system_score_codex":0.0023039829,"about_ca_system_score_gemma":0.0021247368,"threshold_uncertainty_score":0.12174815},"labels":[],"label_agreement":null},{"id":"W4394687825","doi":"10.1002/sam.11679","title":"Data‐driven stochastic model for quantifying the interplay between amyloid‐beta and calcium levels in Alzheimer's disease","year":2024,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Alzheimer's disease research and treatments","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Prince Edward Island; Wilfrid Laurier University; University of Manitoba","funders":"Janssen Alzheimer Immunotherapy Research And Development; Johnson and Johnson Pharmaceutical Research and Development; National Institute on Aging; Agencia Estatal de Investigación; National Institute of Biomedical Imaging and Bioengineering; Canadian Institutes of Health Research; Genentech; National Institutes of Health; Eisai Canada; H. Lundbeck A/S; Servier; Eisai; Shared Hierarchical Academic Research Computing Network; Ministerio de Ciencia, Innovación y Universidades; Northern California Institute for Research and Education; IXICO; Takeda Pharmaceutical Company; Alzheimer's Association; Fujirebio US; DoD Alzheimer's Disease Neuroimaging Initiative; Natural Sciences and Engineering Research Council of Canada; Pfizer; Biogen; BioClinica; F. Hoffmann-La Roche; Roche; AbbVie Canada; University of Southern California; Novartis Pharmaceuticals Corporation; Chesapeake Research Consortium; U.S. Department of Defense; Eli Lilly and Company; Alliance de recherche numérique du Canada; Bristol-Myers Squibb; Alzheimer's Drug Discovery Foundation; Merck; GE Healthcare; Basque Center for Applied Mathematics; Alzheimer's Disease Neuroimaging Initiative; Meso Scale Diagnostics","keywords":"Disease; BETA (programming language); Amyloid (mycology); Amyloid beta; Alzheimer's disease; Neuroscience; Computer science; Biology; Medicine; Pathology","score_opus":0.3030961333940841,"score_gpt":0.48760745701371816,"score_spread":0.18451132361963407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394687825","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2300762,0.0010787532,0.75861245,0.002436841,0.00018579382,0.00016781218,0.0021755293,0.0005428123,0.0047238064],"genre_scores_gemma":[0.9732444,0.0003897914,0.019863708,0.00024019391,0.00007674804,0.0002479426,0.0008999961,0.00005772631,0.0049793734],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991736,0.00028200174,0.00004305464,0.00022396063,0.00014537276,0.00013209833],"domain_scores_gemma":[0.99588996,0.0027266413,0.0005620819,0.00009059468,0.00052161864,0.0002090717],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002905439,0.0012712417,0.0016850302,0.0010078672,0.00050940603,0.0015414174,0.0019850251,0.0027010478,0.00208074],"category_scores_gemma":[0.007288339,0.0010256672,0.0013259238,0.00077296566,0.0015514229,0.0011379919,0.0012518018,0.0017851848,0.00023665781],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026044429,0.000016187618,0.0008785382,0.00001912263,0.000022949718,0.000046682646,0.000014592669,0.9909464,0.00026692828,0.0069090193,0.00019541875,0.00065815187],"study_design_scores_gemma":[0.0000051601323,0.000005729897,0.00014399212,0.0000022733,0.0000047230174,0.000004280328,0.000002594341,0.998221,0.000026122947,0.0015169757,0.00006207913,0.0000050788094],"about_ca_topic_score_codex":0.036325384,"about_ca_topic_score_gemma":0.015658936,"teacher_disagreement_score":0.036325384,"about_ca_system_score_codex":0.002148342,"about_ca_system_score_gemma":0.002186809,"threshold_uncertainty_score":0.072227895},"labels":[],"label_agreement":null},{"id":"W4413289142","doi":"10.1002/sam.70038","title":"Distributionally Conservative Stochastic Dominance via Subsampling","year":2025,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Risk and Portfolio Optimization","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Stochastic dominance; Dominance (genetics); Econometrics; Statistics; Mathematics; Computer science; Economics; Mathematical optimization; Biology","score_opus":0.14582943694385603,"score_gpt":0.4572373035182113,"score_spread":0.31140786657435526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413289142","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024604019,0.0003638171,0.97193074,0.0002849822,0.000039978968,0.00005684038,0.000045716235,0.00007303392,0.0026007134],"genre_scores_gemma":[0.81418496,0.00061099685,0.17979825,0.000648849,0.00026306958,0.0002999432,0.00023208924,0.00014350159,0.0038183571],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9747688,0.015323626,0.00086555164,0.0021251163,0.0060672504,0.0008496195],"domain_scores_gemma":[0.9108527,0.07118758,0.004734435,0.0057321317,0.0063461866,0.0011470327],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.029511552,0.0010755524,0.0018892015,0.0020017046,0.0010785357,0.0030085552,0.0021698156,0.0012709172,0.0021136345],"category_scores_gemma":[0.08344721,0.0006314947,0.001207402,0.0010930096,0.0046030614,0.003058158,0.0043431097,0.0029254258,0.00024008988],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016633882,0.000061009898,0.0024466098,0.00016362198,0.00018219958,0.00031236946,0.00034285,0.18770629,0.003991262,0.7677635,0.0013556094,0.03550827],"study_design_scores_gemma":[0.000015087801,0.00008372888,0.00071490224,0.00003948673,0.000023250139,0.00013654678,0.000048839087,0.7414221,0.002118274,0.253989,0.0013752577,0.000033496486],"about_ca_topic_score_codex":0.0025851242,"about_ca_topic_score_gemma":0.0016135167,"teacher_disagreement_score":0.029511552,"about_ca_system_score_codex":0.0024191283,"about_ca_system_score_gemma":0.0016737387,"threshold_uncertainty_score":0.15607387},"labels":[],"label_agreement":null},{"id":"W4413868413","doi":"10.1002/sam.70042","title":"Recursive Random Binning to Detect and Display Pairwise Dependence","year":2025,"lang":"en","type":"article","venue":"Statistical Analysis and Data Mining The ASA Data Science Journal","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Pairwise comparison; Computer science; Mathematics; Statistics; Algorithm; Artificial intelligence","score_opus":0.04648949380676017,"score_gpt":0.3791487152178348,"score_spread":0.3326592214110746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413868413","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048807394,0.0002611476,0.9450152,0.000105199775,0.00005303582,0.0001910447,0.0007262904,0.0038064462,0.0010342657],"genre_scores_gemma":[0.40425587,0.00012422109,0.59073853,0.00012288983,0.000037315935,0.0005426437,0.0020211963,0.0010116787,0.0011456504],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99561334,0.0025595492,0.00022056731,0.00069755176,0.0006453533,0.00026370067],"domain_scores_gemma":[0.97933334,0.014542919,0.0010060192,0.0031733406,0.001667376,0.00027703203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060113324,0.0006106421,0.0012323986,0.0036466226,0.0005641975,0.0017152776,0.001551522,0.00063953083,0.0077013588],"category_scores_gemma":[0.039122686,0.00041608096,0.0009888846,0.0029975222,0.0013052549,0.0013063436,0.0020155418,0.0015786369,0.0013895114],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016363697,0.00031451136,0.033463772,0.00068900996,0.00058379123,0.00041081072,0.0017611122,0.12372056,0.023245549,0.117239125,0.013288889,0.68364656],"study_design_scores_gemma":[0.00014605252,0.00024358353,0.01514186,0.00012091539,0.00010307191,0.0002654914,0.00037740794,0.831229,0.014999329,0.12523124,0.012026156,0.000115871066],"about_ca_topic_score_codex":0.0036472238,"about_ca_topic_score_gemma":0.003530883,"teacher_disagreement_score":0.0077013588,"about_ca_system_score_codex":0.0008056467,"about_ca_system_score_gemma":0.0011003395,"threshold_uncertainty_score":0.03179139},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"medium"}],"label_agreement":"split"}]}