{"meta":{"query_hash":"210e62fba4a8","filters":{"venue":"Software Testing Verification and Reliability"},"cohort_total":28,"direct_labels_cover":0,"predictions_cover":28,"exported":28,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/210e62fba4a8","api":"https://metacan.xera.ac/api/v1/cohort?venue=Software+Testing+Verification+and+Reliability"},"results":[{"id":"W1516438975","doi":"10.1002/stvr.1573","title":"Anomaly detection in performance regression testing by transaction profile estimation","year":2015,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Science Foundation Ireland","keywords":"Computer science; Regression testing; Workload; Anomaly detection; Data mining; Software performance testing; Software regression; Regression analysis; Regression; Non-regression testing; Software; Machine learning; Software quality; Statistics; Software system; Operating system; Software development","score_opus":0.02938539619239948,"score_gpt":0.2524789475198476,"score_spread":0.22309355132744813,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1516438975","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35053465,0.00037091871,0.6422047,0.0002611843,0.000053323256,0.00016892847,0.00040274175,0.00456121,0.0014423548],"genre_scores_gemma":[0.91775686,0.00007165835,0.081222594,0.000033708522,0.000021005284,0.00007256972,0.00042538342,0.00009236484,0.0003037977],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9947577,0.001945763,0.00040083186,0.0008661965,0.0017585662,0.0002708628],"domain_scores_gemma":[0.9764414,0.0116850445,0.0046122232,0.0029316575,0.0038496992,0.00047991265],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033032058,0.0007919786,0.0007953395,0.0032864863,0.00029453816,0.0011353148,0.0012561034,0.0006345604,0.00062348094],"category_scores_gemma":[0.022220926,0.000328433,0.0005577668,0.0017295643,0.0004239484,0.0012601983,0.00086839264,0.001035357,0.00052969076],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061471626,0.0004365695,0.22922745,0.00023100484,0.00021829267,0.00070191757,0.00047166215,0.13246268,0.04910176,0.0031211036,0.0029534844,0.5804594],"study_design_scores_gemma":[0.000009377711,0.00011241717,0.014446481,0.000016448852,0.0000213828,0.0003209793,0.00007030769,0.97154343,0.010710632,0.0021305725,0.00059516507,0.000022796587],"about_ca_topic_score_codex":0.0028788573,"about_ca_topic_score_gemma":0.0021572157,"teacher_disagreement_score":0.0033032058,"about_ca_system_score_codex":0.00053057144,"about_ca_system_score_gemma":0.000659874,"threshold_uncertainty_score":0.017469227},"labels":[],"label_agreement":null},{"id":"W1522741019","doi":"10.1002/stvr.1572","title":"Coverage‐based regression test case selection, minimization and prioritization: a case study on an industrial system","year":2015,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":95,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg","keywords":"Regression testing; Minification; Selection (genetic algorithm); Computer science; Prioritization; Fault detection and isolation; Suite; Reliability engineering; Test suite; Regression; Fault (geology); Regression analysis; Data mining; Machine learning; Test case; Artificial intelligence; Statistics; Engineering; Software; Mathematics; Software system","score_opus":0.08433280682020698,"score_gpt":0.31182946332980477,"score_spread":0.22749665650959777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1522741019","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9730856,0.00013677601,0.02413686,0.00019162367,0.00000695634,0.00016013047,0.00007457532,0.00026397375,0.0019434575],"genre_scores_gemma":[0.97401583,0.000071771894,0.02511113,0.000025665748,0.0000045079405,0.00005683756,0.000075129974,0.000029043966,0.0006099965],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977835,0.0010923035,0.0001004659,0.0001448867,0.0007050007,0.00017380404],"domain_scores_gemma":[0.985876,0.011496888,0.00071484956,0.0006826772,0.0010271533,0.00020235786],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023628203,0.00047692825,0.00037538438,0.0011631858,0.0005224155,0.00047363777,0.0009475021,0.0007806709,0.0010363775],"category_scores_gemma":[0.009054066,0.00026103697,0.00043809277,0.0008853189,0.0005641717,0.0004912508,0.00038070767,0.0004981832,0.00014136282],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016223679,0.0025813682,0.05502882,0.0008289422,0.00023156728,0.010983381,0.0018961363,0.58164716,0.06314454,0.0049145604,0.0029322458,0.2741889],"study_design_scores_gemma":[0.000512466,0.0038626858,0.029809136,0.000075340235,0.00022277806,0.0036294411,0.0010607197,0.8720822,0.08189059,0.0016455263,0.00513286,0.00007624548],"about_ca_topic_score_codex":0.007404279,"about_ca_topic_score_gemma":0.008795127,"teacher_disagreement_score":0.007404279,"about_ca_system_score_codex":0.0009127594,"about_ca_system_score_gemma":0.00076461985,"threshold_uncertainty_score":0.014722347},"labels":[],"label_agreement":null},{"id":"W1527815941","doi":"10.1002/stvr.1576","title":"Automatic fault localization for client‐side JavaScript","year":2015,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Intel Corporation","keywords":"JavaScript; Unobtrusive JavaScript; Computer science; Ajax; Scripting language; Web application; Debugging; Programming language; Rich Internet application; Document Object Model; Code (set theory); Source code; Dynamic web page; Tracing; Operating system; World Wide Web; Web service; Web page","score_opus":0.05771615534496858,"score_gpt":0.29183010284763583,"score_spread":0.23411394750266726,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1527815941","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19043401,0.000654395,0.7327089,0.00020089731,0.000070735645,0.00012118957,0.00027044286,0.07363682,0.0019026608],"genre_scores_gemma":[0.8022486,0.00012168762,0.19386353,0.00007133304,0.000019596748,0.0000553434,0.0006019349,0.0013279072,0.0016900374],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99803585,0.00041148937,0.00013501669,0.0004109276,0.00087246316,0.00013430754],"domain_scores_gemma":[0.9923896,0.002901183,0.0011705635,0.0015311972,0.0018293151,0.00017812384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009757125,0.0009826102,0.00072305615,0.0018284121,0.0004180492,0.0007723214,0.0013455887,0.0007999582,0.001384773],"category_scores_gemma":[0.005944906,0.0004004349,0.00057361746,0.0005100758,0.0005181585,0.00085974886,0.00081776566,0.00073483086,0.0009633389],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046624022,0.0002354799,0.022451712,0.0005634383,0.00009094829,0.00093860226,0.00066442485,0.07010558,0.19205551,0.0032375741,0.007683363,0.70150703],"study_design_scores_gemma":[0.0000352618,0.00016769748,0.0058811023,0.00006868567,0.000043155644,0.00085647526,0.00008698007,0.82660633,0.1585603,0.00305994,0.004590109,0.000043889795],"about_ca_topic_score_codex":0.0027876534,"about_ca_topic_score_gemma":0.0019579085,"teacher_disagreement_score":0.0027876534,"about_ca_system_score_codex":0.000668508,"about_ca_system_score_gemma":0.0010163786,"threshold_uncertainty_score":0.0055428743},"labels":[],"label_agreement":null},{"id":"W1969864380","doi":"10.1002/stvr.241","title":"Editorial: <i>Mutation 2000—A Symposium on Mutation Testing</i>","year":2001,"lang":"en","type":"editorial","venue":"Software Testing Verification and Reliability","topic":"Genetics, Bioinformatics, and Biomedical Research","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Bell (Canada)","funders":"","keywords":"Citation; Mutation; Library science; Computer science; World Wide Web; Information retrieval; Genetics; Biology","score_opus":0.014995523560344589,"score_gpt":0.281196696684783,"score_spread":0.26620117312443836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1969864380","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00003786473,0.0046356,0.00020085802,0.043121547,0.9503774,0.00002634199,0.000051209678,0.00007496066,0.0014741969],"genre_scores_gemma":[0.00057024945,0.0037509652,0.00024155037,0.05368655,0.92883307,0.000050660878,0.000080629005,0.000090956644,0.012695387],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9948191,0.0008516374,0.0006550182,0.0006046344,0.0026800123,0.00038963926],"domain_scores_gemma":[0.97441125,0.009155212,0.0019258227,0.0007147852,0.010044931,0.0037480474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00942223,0.004981894,0.004633367,0.005112043,0.0039007918,0.008609824,0.0042916504,0.018442824,0.011383573],"category_scores_gemma":[0.0269629,0.0015218222,0.0037172325,0.002228325,0.0036702086,0.004852087,0.0018169674,0.029214371,0.009834596],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037111822,0.000008962589,0.000013648134,0.000071535665,0.0000142637955,0.00005367148,0.0000044018725,0.000016147867,0.000036552945,0.000115681454,0.9972947,0.0023332278],"study_design_scores_gemma":[0.00012497387,0.000050935534,0.00037873906,0.000459119,0.00010816444,0.00024717257,0.00003431208,0.0002493284,0.00018830367,0.0010735218,0.99705374,0.000031696345],"about_ca_topic_score_codex":0.0024777774,"about_ca_topic_score_gemma":0.0072404747,"teacher_disagreement_score":0.018442824,"about_ca_system_score_codex":0.0031114435,"about_ca_system_score_gemma":0.0027525339,"threshold_uncertainty_score":0.04983008},"labels":[],"label_agreement":null},{"id":"W1982395839","doi":"10.1002/1099-1689(200009)10:3<149::aid-stvr206>3.0.co;2-t","title":"State generation and automated class testing","year":2000,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Container (type theory); Computer science; White-box testing; Tree (set theory); Class (philosophy); Block (permutation group theory); Java; Reliability (semiconductor); Black box; Code coverage; Reliability engineering; Data mining; Programming language; Engineering; Artificial intelligence; Software; Software development","score_opus":0.03457682983096203,"score_gpt":0.2649357170513301,"score_spread":0.2303588872203681,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982395839","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02540016,0.000054070206,0.96835977,0.00008360656,0.000015593425,0.00010705565,0.00006170208,0.0037406415,0.002177354],"genre_scores_gemma":[0.360162,0.00012209876,0.6353642,0.00010598236,0.000019416097,0.000350787,0.00057307124,0.0007585954,0.0025439286],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99756515,0.0009266269,0.00012594827,0.00039912565,0.00081143965,0.00017178302],"domain_scores_gemma":[0.9918333,0.0054689124,0.0007414915,0.0012179198,0.0006404644,0.00009786544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017572932,0.0008777852,0.00069743465,0.0013587369,0.00046451512,0.0012780859,0.0015793766,0.0008086043,0.004091278],"category_scores_gemma":[0.010765,0.00043681465,0.00072004955,0.00090052735,0.0011911046,0.0018529606,0.0011173487,0.0007587568,0.0010063564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003560525,0.00025513454,0.0038520365,0.00024637248,0.000053177824,0.00035757152,0.0003510965,0.22275989,0.02737512,0.08643515,0.004839147,0.65311927],"study_design_scores_gemma":[0.00007693913,0.00012607595,0.00064371957,0.00003073677,0.000021718159,0.00023840189,0.000035382865,0.9267737,0.029762425,0.037156396,0.005107242,0.000027211905],"about_ca_topic_score_codex":0.0016801618,"about_ca_topic_score_gemma":0.0015235969,"teacher_disagreement_score":0.004091278,"about_ca_system_score_codex":0.0010525269,"about_ca_system_score_gemma":0.0011519399,"threshold_uncertainty_score":0.013686717},"labels":[],"label_agreement":null},{"id":"W2035512953","doi":"10.1002/stvr.418","title":"Fault‐driven stress testing of distributed real‐time software based on UML models","year":2009,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Stress testing (software); Unified Modeling Language; Test case; Test (biology); Stress test; Node (physics); Stress (linguistics); Cover (algebra); Real-time computing; Software; Reliability engineering; Programming language; Engineering; Machine learning","score_opus":0.02393212963172253,"score_gpt":0.24975384158457228,"score_spread":0.22582171195284975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2035512953","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24042906,0.0002383296,0.7557371,0.00016249194,0.000030078918,0.00010310292,0.00006123349,0.0020299437,0.0012086171],"genre_scores_gemma":[0.9218067,0.000072161434,0.07751249,0.000029835088,0.000009255921,0.0000908719,0.00006787291,0.00004613912,0.0003647525],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978369,0.0009972958,0.00011600083,0.0001926325,0.0007474731,0.000109692555],"domain_scores_gemma":[0.9921136,0.004486804,0.0010494838,0.00113182,0.0010640713,0.00015421897],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017080827,0.0007678757,0.0003629651,0.0009178503,0.00018658444,0.0006863951,0.0010597808,0.0006811529,0.00069718977],"category_scores_gemma":[0.006985927,0.00029837235,0.00062347564,0.00025716895,0.00073497725,0.0009484143,0.0005686816,0.00047274065,0.000105624385],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054033886,0.00031336796,0.011696173,0.00028428627,0.00015068504,0.000980061,0.0006917099,0.75322306,0.09476797,0.023370244,0.0006958617,0.113286294],"study_design_scores_gemma":[0.000033561548,0.00012945528,0.00068455655,0.000021277143,0.000022811437,0.000116014424,0.000023190516,0.9719492,0.02127809,0.005005062,0.000725316,0.000011398677],"about_ca_topic_score_codex":0.0019844472,"about_ca_topic_score_gemma":0.0016624294,"teacher_disagreement_score":0.0019844472,"about_ca_system_score_codex":0.0007927423,"about_ca_system_score_gemma":0.0007789877,"threshold_uncertainty_score":0.009033263},"labels":[],"label_agreement":null},{"id":"W2053464286","doi":"10.1002/stvr.1476","title":"Status and Awards","year":2012,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Labor Movements and Unions","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Business","score_opus":0.03552312542783721,"score_gpt":0.314535663699376,"score_spread":0.2790125382715388,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053464286","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011128935,0.010379023,0.002200796,0.1795644,0.5547177,0.0005000011,0.002093273,0.0011003156,0.24833156],"genre_scores_gemma":[0.0057312697,0.005203906,0.0017036236,0.033738144,0.048928984,0.0004784291,0.0027479744,0.00081267627,0.9006551],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98441964,0.0014369785,0.0006645899,0.0014340346,0.0095654335,0.0024792915],"domain_scores_gemma":[0.9464229,0.0024069822,0.0013572928,0.0022485333,0.024649983,0.022914378],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.012482761,0.0013455534,0.0018372096,0.0034393,0.005693255,0.014441967,0.0050414493,0.008803275,0.36048022],"category_scores_gemma":[0.04519279,0.00058909593,0.0013082512,0.0026777736,0.0016751926,0.006997653,0.0075416854,0.0074587157,0.2688191],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029465433,0.000037244317,0.000114271796,0.000049218044,0.000002770748,0.000025587491,0.000020259391,0.000018767829,0.00004142583,0.0025796,0.9671108,0.02997055],"study_design_scores_gemma":[0.000011928469,0.000032098254,0.0002466516,0.000078995035,0.0000027374501,0.000033966724,0.0000739181,0.00004404878,0.000042746586,0.0011254002,0.99829966,0.00000786588],"about_ca_topic_score_codex":0.0028047387,"about_ca_topic_score_gemma":0.004295656,"teacher_disagreement_score":0.36048022,"about_ca_system_score_codex":0.003973257,"about_ca_system_score_gemma":0.011868563,"threshold_uncertainty_score":0.91219735},"labels":[],"label_agreement":null},{"id":"W2084111010","doi":"10.1002/stvr.458","title":"An approach for testing pointcut descriptors in AspectJ","year":2011,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"RealNetworks (Canada)","funders":"","keywords":"AspectJ; Computer science; Aspect-oriented programming; Modularity (biology); Set (abstract data type); Programming language; Compile time; Compiler; Software engineering; Base (topology); Test case; Software","score_opus":0.13691961179041653,"score_gpt":0.29543681318094034,"score_spread":0.1585172013905238,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084111010","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11832905,0.00022274414,0.85187906,0.00017534565,0.00009236464,0.0005280019,0.0002108529,0.024878604,0.0036840253],"genre_scores_gemma":[0.48941424,0.0001447932,0.5053358,0.00020492678,0.00004291994,0.00031641472,0.00066871924,0.002061921,0.0018101764],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.994276,0.0011871603,0.00047755285,0.0010303193,0.0026252987,0.00040379594],"domain_scores_gemma":[0.99133474,0.003477745,0.0011220396,0.0019917907,0.0017887034,0.000284961],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031793872,0.00082918175,0.00057105924,0.0022689353,0.0005271341,0.0018223147,0.0023422765,0.0016421973,0.0014915968],"category_scores_gemma":[0.009680127,0.00094321655,0.0010193208,0.00095931214,0.0015741661,0.0025454878,0.0018434111,0.0016258748,0.00052251184],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017052107,0.001954918,0.06507691,0.0010795379,0.00026289205,0.0035400155,0.0027328539,0.04133151,0.36908063,0.055999104,0.007039491,0.45019695],"study_design_scores_gemma":[0.00034175414,0.0020957992,0.014174925,0.00028434896,0.000305201,0.0029865662,0.0004451898,0.5251177,0.37716046,0.033984084,0.042822797,0.00028115575],"about_ca_topic_score_codex":0.0022355258,"about_ca_topic_score_gemma":0.001776204,"teacher_disagreement_score":0.0031793872,"about_ca_system_score_codex":0.0006323656,"about_ca_system_score_gemma":0.0012495178,"threshold_uncertainty_score":0.01681441},"labels":[],"label_agreement":null},{"id":"W2087103218","doi":"10.1002/stvr.295","title":"Eight maxims for software inspectors","year":2004,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Cover (algebra); Set (abstract data type); Software; Engineering; Engineering ethics; Software inspection; Computer science; Software engineering; Management science; Software development; Software quality; Mechanical engineering","score_opus":0.027674755534598493,"score_gpt":0.2663009924304416,"score_spread":0.23862623689584309,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2087103218","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41460073,0.0053597568,0.08686528,0.36885592,0.008482774,0.0012471373,0.00026186477,0.0010283282,0.11329813],"genre_scores_gemma":[0.88761246,0.0015424051,0.061435606,0.015878728,0.001095999,0.00076068117,0.0002054803,0.00030582683,0.031162774],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.95807534,0.025160061,0.0018576377,0.0017728424,0.010401082,0.0027329994],"domain_scores_gemma":[0.93964684,0.029268987,0.005105658,0.0034591183,0.0147668645,0.007752481],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.035414256,0.0009591367,0.0006887206,0.0017361244,0.006588692,0.00840077,0.0019541848,0.005078459,0.0035154682],"category_scores_gemma":[0.06648682,0.0007391477,0.0006764373,0.0014140205,0.009756284,0.0074363225,0.010510007,0.012835501,0.0011710202],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045583775,0.00043578213,0.011273676,0.0014300621,0.000066848545,0.0011329058,0.37141737,0.0009873825,0.014794539,0.23034765,0.14261681,0.225041],"study_design_scores_gemma":[0.00005352048,0.00057184254,0.011391728,0.0016821345,0.00003133688,0.0010339426,0.40896744,0.003323673,0.0024187374,0.14287305,0.42743123,0.00022135398],"about_ca_topic_score_codex":0.00053900905,"about_ca_topic_score_gemma":0.0007658931,"teacher_disagreement_score":0.035414256,"about_ca_system_score_codex":0.006582958,"about_ca_system_score_gemma":0.005406913,"threshold_uncertainty_score":0.18729079},"labels":[],"label_agreement":null},{"id":"W2111584600","doi":"10.1002/stvr.452","title":"On reducing test length for FSMs with extra states","year":2011,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Tree traversal; Test suite; Computer science; Reduction (mathematics); Set (abstract data type); Algorithm; Finite-state machine; Test set; Implementation; Test case; Theoretical computer science; Mathematics; Programming language; Artificial intelligence","score_opus":0.04516024558319632,"score_gpt":0.25357394988629844,"score_spread":0.20841370430310213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111584600","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26682436,0.00041506076,0.72593534,0.00067201996,0.00006286509,0.00021829447,0.0001364565,0.0032199742,0.002515689],"genre_scores_gemma":[0.64319766,0.00020016685,0.35317352,0.00024218226,0.00009196489,0.00031532452,0.0005145087,0.00051600696,0.0017486374],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99669266,0.0013726215,0.00020033073,0.00034770306,0.0010663036,0.00032042695],"domain_scores_gemma":[0.9642112,0.02630382,0.0025760487,0.0042779506,0.001989783,0.000641047],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021143015,0.00079936493,0.0010178074,0.0010861047,0.0006148038,0.0006593789,0.0016600672,0.0006184493,0.0025915424],"category_scores_gemma":[0.019881576,0.0005111171,0.0010626302,0.00087257125,0.0009857932,0.0025368833,0.0016954124,0.0019228149,0.00037875524],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022940165,0.0005101995,0.0057061682,0.00042544448,0.00013806642,0.0004867791,0.00038125852,0.3548139,0.0684255,0.019762386,0.0030203927,0.5440359],"study_design_scores_gemma":[0.00018369283,0.0010926074,0.0017275421,0.000049481765,0.00007101305,0.00028761823,0.000080660946,0.9344681,0.03316031,0.026220765,0.0026311604,0.000027034246],"about_ca_topic_score_codex":0.0014043475,"about_ca_topic_score_gemma":0.002185384,"teacher_disagreement_score":0.0025915424,"about_ca_system_score_codex":0.0011278823,"about_ca_system_score_gemma":0.0015171719,"threshold_uncertainty_score":0.011181593},"labels":[],"label_agreement":null},{"id":"W2145149095","doi":"10.1002/stvr.461","title":"Regression test suite prioritization using system models","year":2011,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Regression testing; Test suite; Computer science; Prioritization; Reliability engineering; Test (biology); Test case; Fault detection and isolation; System under test; Test Management Approach; Empirical research; Overhead (engineering); Suite; Regression analysis; Data mining; Machine learning; Artificial intelligence; Engineering; Software system; Statistics; Software; Programming language","score_opus":0.08413484429999112,"score_gpt":0.2648815431189903,"score_spread":0.18074669881899919,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145149095","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19735795,0.00039249365,0.7927605,0.00031141329,0.000032935473,0.00027071693,0.00019602191,0.0041514584,0.004526441],"genre_scores_gemma":[0.83300287,0.00011645614,0.16547835,0.000041131912,0.000016906086,0.00012778513,0.00025626228,0.00022134294,0.0007389513],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9939166,0.0034334392,0.00023685349,0.00042312787,0.0017089498,0.00028115758],"domain_scores_gemma":[0.9773501,0.016708946,0.0016827426,0.0019348817,0.0020657266,0.00025752344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005257537,0.0010585084,0.0007414755,0.0021619361,0.00027091222,0.0010771577,0.0011154748,0.0004625328,0.001777706],"category_scores_gemma":[0.021949288,0.0004777441,0.0007222638,0.001081807,0.00039705503,0.0013267562,0.0008196723,0.00097353256,0.0002456773],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004814765,0.0002798837,0.009404477,0.00015791623,0.00011559735,0.00011606793,0.00013761055,0.7876969,0.009746252,0.009891013,0.0015090597,0.18046382],"study_design_scores_gemma":[0.000025894271,0.00009559602,0.0006630533,0.000010279412,0.00002499352,0.000037146205,0.000011558931,0.99200165,0.0040802574,0.0025402976,0.00049831445,0.000010843089],"about_ca_topic_score_codex":0.0051153935,"about_ca_topic_score_gemma":0.0051773717,"teacher_disagreement_score":0.005257537,"about_ca_system_score_codex":0.0014784358,"about_ca_system_score_gemma":0.0014399134,"threshold_uncertainty_score":0.027804792},"labels":[],"label_agreement":null},{"id":"W2155388066","doi":"10.1002/stvr.210","title":"A rigorous method for test templates generation from object‐oriented specifications","year":2001,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Template; Computer science; Formal specification; Programming language; Focus (optics); Test case; White-box testing; Object-oriented programming; Formal methods; Extension (predicate logic); Specification language; Test (biology); Software engineering; Software; Software development; Software construction","score_opus":0.07291649913695267,"score_gpt":0.30554752011573033,"score_spread":0.23263102097877766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155388066","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017633445,0.000015252163,0.99673504,0.000035092544,0.0000102821905,0.00015424647,0.00002140758,0.0008371166,0.00042828353],"genre_scores_gemma":[0.05179903,0.00005852192,0.94561505,0.00007339754,0.00001716187,0.0005720021,0.00030080136,0.0005379662,0.001026079],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9828483,0.006471434,0.0011336352,0.0011561452,0.007883531,0.0005069152],"domain_scores_gemma":[0.96016157,0.025139192,0.001987503,0.007271876,0.0049925568,0.00044731543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010735871,0.0011868536,0.001089416,0.0025812823,0.00074160344,0.002095854,0.0025260383,0.0014191268,0.004325801],"category_scores_gemma":[0.046297334,0.0012014214,0.0019062845,0.0010249731,0.003292832,0.0019435106,0.0032068472,0.002244923,0.0016061929],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036051712,0.00047583948,0.002353314,0.0008101836,0.0002248219,0.0010773506,0.0017699623,0.09462225,0.06378303,0.36461055,0.0058782967,0.4640339],"study_design_scores_gemma":[0.0005520757,0.0005353374,0.00068121427,0.0003185518,0.00013362349,0.0009978855,0.00022683475,0.6496534,0.10864464,0.2006394,0.03745323,0.00016387136],"about_ca_topic_score_codex":0.0008536291,"about_ca_topic_score_gemma":0.0006829304,"teacher_disagreement_score":0.010735871,"about_ca_system_score_codex":0.00093601743,"about_ca_system_score_gemma":0.0027257216,"threshold_uncertainty_score":0.056777358},"labels":[],"label_agreement":null},{"id":"W2196997841","doi":"","title":"Eight maxims for software inspectors: Research Articles","year":2004,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Cover (algebra); Set (abstract data type); Software; Software inspection; Engineering; Engineering ethics; Computer science; Software engineering; Engineering management; Sociology; Software development; Software quality; Mechanical engineering","score_opus":0.07757743155948835,"score_gpt":0.3266650500084117,"score_spread":0.24908761844892333,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2196997841","genre_codex":"review","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017194394,0.48985675,0.009505425,0.37939814,0.02922897,0.0007962886,0.0005927523,0.00026510408,0.07316216],"genre_scores_gemma":[0.16380355,0.63740927,0.032941915,0.055591222,0.040801983,0.0024987538,0.001929757,0.00049348304,0.06453008],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9827717,0.0064336834,0.0025886616,0.00091175875,0.0065327934,0.00076139404],"domain_scores_gemma":[0.8497234,0.10649502,0.009047345,0.0027170433,0.025712345,0.006304829],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023365673,0.0010385851,0.0011484043,0.011748913,0.0040920186,0.01914377,0.001456853,0.0058302903,0.005629617],"category_scores_gemma":[0.09638973,0.0008687281,0.00060178747,0.017830074,0.005342924,0.01652768,0.00512289,0.005475659,0.0023098015],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026520228,0.00033525765,0.0019027079,0.017143019,0.000073345356,0.0003784376,0.02239866,0.00032908388,0.001793244,0.094792105,0.38674968,0.47383928],"study_design_scores_gemma":[0.000049930186,0.00016679673,0.0034445687,0.025068585,0.000079241625,0.00045676876,0.0332283,0.00022141992,0.0006697369,0.037262708,0.899288,0.00006398183],"about_ca_topic_score_codex":0.0006853245,"about_ca_topic_score_gemma":0.0010858141,"teacher_disagreement_score":0.023365673,"about_ca_system_score_codex":0.0061269873,"about_ca_system_score_gemma":0.0074224295,"threshold_uncertainty_score":0.12357098},"labels":[],"label_agreement":null},{"id":"W2461407631","doi":"10.1002/stvr.1609","title":"Prioritizing manual test cases in rapid release environments","year":2016,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Manitoba","funders":"Mitacs; Lunds Universitet; University of Manitoba","keywords":"Computer science; Prioritization; Test suite; Unit testing; Agile software development; Test (biology); Code coverage; Suite; Test case; Embedded system; Software engineering; Operating system; Software; Engineering; Machine learning","score_opus":0.02512928207923182,"score_gpt":0.25991368093417877,"score_spread":0.23478439885494695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2461407631","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6302709,0.0010893104,0.34764308,0.0005501575,0.00012466336,0.0010467428,0.00025773342,0.008609787,0.010407738],"genre_scores_gemma":[0.8363888,0.0002539735,0.15883611,0.00014934575,0.000057339606,0.00026178895,0.0005662239,0.00064382725,0.0028424999],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9859087,0.005914173,0.0007183689,0.0014201937,0.0051843156,0.0008542521],"domain_scores_gemma":[0.9203357,0.055746738,0.008115263,0.007098893,0.0068033915,0.001899988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068255737,0.0013188218,0.0007074249,0.0033435458,0.00043743447,0.0018425514,0.0021106596,0.0006016444,0.0029470662],"category_scores_gemma":[0.035307616,0.00072936394,0.00059170637,0.0009803995,0.0005939524,0.0016589187,0.0017615611,0.0011154562,0.001017981],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018437615,0.0015501188,0.024738155,0.00077561877,0.0001675085,0.0016581807,0.0018571629,0.0584857,0.07948006,0.0036011073,0.005904751,0.8199379],"study_design_scores_gemma":[0.0012018284,0.009979488,0.08887967,0.0008304685,0.0005010291,0.0063994657,0.0032336428,0.57981807,0.24085876,0.015683284,0.052048925,0.0005654374],"about_ca_topic_score_codex":0.0017800006,"about_ca_topic_score_gemma":0.0022011853,"teacher_disagreement_score":0.0068255737,"about_ca_system_score_codex":0.0006902158,"about_ca_system_score_gemma":0.0011616145,"threshold_uncertainty_score":0.036097527},"labels":[],"label_agreement":null},{"id":"W2806282111","doi":"10.1002/stvr.1669","title":"MuMonDE: A framework for evaluating model clone detectors using model mutation analysis","year":2018,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Engineering Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"clone (Java method); Preprocessor; Mutation; Computer science; Mutation testing; Data mining; Detector; Precision and recall; Software engineering; Artificial intelligence; Genetics; Biology","score_opus":0.10310354306044087,"score_gpt":0.3755641998763948,"score_spread":0.2724606568159539,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806282111","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012272797,0.00009993945,0.9698473,0.00014300506,0.000021504104,0.00039591824,0.00042035713,0.015280713,0.0015184588],"genre_scores_gemma":[0.11326995,0.00006171018,0.8836408,0.000075488424,0.000014664864,0.00054187345,0.00094979943,0.00090699404,0.00053873623],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98712677,0.004957865,0.0014800511,0.0012855572,0.004701756,0.00044800248],"domain_scores_gemma":[0.96620876,0.020126276,0.0033394685,0.0043413965,0.0053999186,0.0005841756],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.017811961,0.0022560563,0.0014334898,0.0093310755,0.00089610304,0.004315079,0.0038385778,0.0014435882,0.0040845787],"category_scores_gemma":[0.056034375,0.0011020115,0.0025389884,0.0020239325,0.0020449795,0.003669194,0.0038627102,0.0019126841,0.0007796316],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00080359966,0.0009598485,0.02600587,0.0011385577,0.0006939824,0.00050268177,0.001260492,0.39427823,0.025201248,0.124106735,0.0130271325,0.41202164],"study_design_scores_gemma":[0.000070077724,0.00020322403,0.0014997103,0.00013754432,0.00006534411,0.00013114873,0.00011178276,0.9632713,0.011154599,0.017317314,0.0059581664,0.00007977167],"about_ca_topic_score_codex":0.008010807,"about_ca_topic_score_gemma":0.008931389,"teacher_disagreement_score":0.98218805,"about_ca_system_score_codex":0.0027104998,"about_ca_system_score_gemma":0.002975256,"threshold_uncertainty_score":0.094199836},"labels":[],"label_agreement":null},{"id":"W2809981234","doi":"10.1002/stvr.1665","title":"P<scp>esto</scp>: Automated migration of DOM‐based Web tests towards the visual approach","year":2018,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Automation; Web application; Test (biology); Visual Basic; Visualization; Software engineering; Artificial intelligence; Programming language; Software; World Wide Web; Engineering","score_opus":0.033805008270209574,"score_gpt":0.28879608613740154,"score_spread":0.254991077867192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2809981234","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06062366,0.000089150715,0.7077764,0.00027024795,0.000114356335,0.00033563966,0.0015813595,0.2209654,0.008243709],"genre_scores_gemma":[0.5276784,0.00012227561,0.43895453,0.00042318678,0.000050951156,0.0005160249,0.0064631877,0.016246004,0.009545411],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99817777,0.0004663448,0.0002186325,0.00029772153,0.0006835867,0.00015600333],"domain_scores_gemma":[0.9923086,0.0023844915,0.00062668824,0.0029460308,0.0014284017,0.00030577672],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017495062,0.0012307152,0.00055310037,0.001747122,0.000417167,0.0014616555,0.0016849074,0.00091167685,0.0050389194],"category_scores_gemma":[0.0080501335,0.00062432955,0.00069003657,0.00059765653,0.0007978069,0.0014213177,0.001554424,0.0010861006,0.002631515],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001383808,0.00071207137,0.021548178,0.00068080006,0.00017873036,0.0037191007,0.0007477165,0.05693682,0.13130903,0.011082459,0.10386644,0.6678348],"study_design_scores_gemma":[0.00025541216,0.0004348807,0.0111313425,0.0001604301,0.000059802744,0.0018636078,0.00013430798,0.69050306,0.22929154,0.006944518,0.059066087,0.00015510188],"about_ca_topic_score_codex":0.002597894,"about_ca_topic_score_gemma":0.0020345333,"teacher_disagreement_score":0.0050389194,"about_ca_system_score_codex":0.0004294301,"about_ca_system_score_gemma":0.0010916257,"threshold_uncertainty_score":0.01685685},"labels":[],"label_agreement":null},{"id":"W2979327180","doi":"10.1002/stvr.1713","title":"An end‐user‐centric test generation methodology for performance evaluation of mobile networked applications","year":2019,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Computer network; Test suite; Wireless network; End user; Wireless; Test case; Machine learning; Operating system","score_opus":0.06774374964768788,"score_gpt":0.3249485911742777,"score_spread":0.25720484152658984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2979327180","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06892248,0.00012172968,0.92709225,0.00013791135,0.000012908713,0.00025769213,0.00010449633,0.0020337251,0.0013167415],"genre_scores_gemma":[0.72349024,0.00006669631,0.27525324,0.000081545346,0.000009284991,0.00023985986,0.0002603031,0.0001286645,0.0004701657],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99612445,0.0018421598,0.00022466939,0.00034475254,0.0012778656,0.00018609573],"domain_scores_gemma":[0.99460065,0.0023726758,0.00063257833,0.0010000404,0.0012862828,0.00010779445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027289586,0.00074969436,0.0004878155,0.0012016932,0.00019815793,0.00070577004,0.0012725694,0.00055940077,0.0009974403],"category_scores_gemma":[0.007976584,0.00029729647,0.0007197003,0.0004231338,0.0005993576,0.0006663869,0.00062255765,0.00062332477,0.0001722477],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019526355,0.00043079362,0.007915697,0.00017052042,0.00011666316,0.00024982856,0.00009273113,0.77564645,0.042787142,0.009332652,0.0011864643,0.16187575],"study_design_scores_gemma":[0.000015988588,0.00021319419,0.00056405884,0.000014779851,0.000017192788,0.00006781422,0.000008220477,0.9868174,0.010532863,0.0013677069,0.00037205635,0.000008768571],"about_ca_topic_score_codex":0.002002022,"about_ca_topic_score_gemma":0.0016427838,"teacher_disagreement_score":0.0027289586,"about_ca_system_score_codex":0.0010385627,"about_ca_system_score_gemma":0.0011967922,"threshold_uncertainty_score":0.014432251},"labels":[],"label_agreement":null},{"id":"W2994711059","doi":"10.1002/stvr.1721","title":"Leveraging metamorphic testing to automatically detect inconsistencies in code generator families","year":2019,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Agence Nationale de la Recherche","keywords":"Computer science; Software quality; Oracle; Software; Leverage (statistics); Regression testing; Code (set theory); Code coverage; Unreachable code; Software development; Code generation; Programming language; Redundant code; Software construction; Set (abstract data type); Operating system; Artificial intelligence","score_opus":0.03641849049665075,"score_gpt":0.2574023251371866,"score_spread":0.22098383464053584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2994711059","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4932825,0.00028323138,0.49618748,0.00016575487,0.000024408562,0.00015069601,0.0002188367,0.008405702,0.0012813088],"genre_scores_gemma":[0.87813514,0.00005324172,0.120824754,0.000057921567,0.000012926441,0.00006927011,0.00038745799,0.00020660601,0.00025281784],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964659,0.0009975854,0.00031637365,0.0008288821,0.0012069517,0.00018424704],"domain_scores_gemma":[0.976344,0.012546669,0.005570724,0.0024820275,0.0026740828,0.0003825188],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002627858,0.00053986127,0.0006672531,0.0037166926,0.00030464842,0.00086493,0.0011994895,0.00063852046,0.00053502194],"category_scores_gemma":[0.017690634,0.0003904501,0.0005939456,0.0010644413,0.0007113995,0.0008786203,0.0011354052,0.00067203795,0.00020296278],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007035004,0.0004319251,0.21237373,0.0003667823,0.0002590825,0.001517168,0.00077849126,0.17012355,0.11569347,0.0084612295,0.0022529871,0.4870381],"study_design_scores_gemma":[0.000029660749,0.00022270545,0.016867412,0.000045015655,0.000044968758,0.0006775658,0.000054094828,0.9500443,0.026262224,0.004777249,0.00094486127,0.000029917108],"about_ca_topic_score_codex":0.0011428174,"about_ca_topic_score_gemma":0.0012762825,"teacher_disagreement_score":0.0037166926,"about_ca_system_score_codex":0.00045127323,"about_ca_system_score_gemma":0.00065874326,"threshold_uncertainty_score":0.013897657},"labels":[],"label_agreement":null},{"id":"W3013655954","doi":"10.1002/stvr.380","title":"Automated discovery of state transitions and their functions in source code","year":2007,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Engineering and Physical Sciences Research Council; National Aeronautics and Space Administration","keywords":"Computer science; State (computer science); Source code; Finite-state machine; Reverse engineering; Set (abstract data type); Code (set theory); Programming language; Software; Software engineering; Transition (genetics); Theoretical computer science","score_opus":0.01937952920259313,"score_gpt":0.2572724153819903,"score_spread":0.2378928861793972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3013655954","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23582442,0.00018941658,0.7460938,0.00017431524,0.00002270662,0.00010429418,0.00018778002,0.015963182,0.001440115],"genre_scores_gemma":[0.75985944,0.000121943776,0.23805591,0.000027250184,0.000006801225,0.00007062204,0.00044467676,0.00069743075,0.0007159485],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983682,0.0005929438,0.00010055019,0.00022591215,0.00060606166,0.00010638159],"domain_scores_gemma":[0.98507,0.010435237,0.0017341479,0.0017193288,0.00094260246,0.00009865344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015079489,0.0006328678,0.00045221113,0.0014044484,0.00040591977,0.0010437056,0.0010110466,0.0008669998,0.0013026537],"category_scores_gemma":[0.013248674,0.0005962955,0.00055026927,0.0005948549,0.001234705,0.0012705142,0.00080484786,0.000954599,0.0004285913],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008378351,0.0003854986,0.030090868,0.0008114832,0.00014064476,0.0023185979,0.0020776694,0.20615298,0.18958293,0.059655555,0.0024098088,0.5055362],"study_design_scores_gemma":[0.00003278826,0.000093572766,0.002062293,0.00005742852,0.000041078456,0.00043209133,0.000058243168,0.86467665,0.11316794,0.016954139,0.0023907886,0.000032923934],"about_ca_topic_score_codex":0.00188932,"about_ca_topic_score_gemma":0.0015839919,"teacher_disagreement_score":0.00188932,"about_ca_system_score_codex":0.0005454903,"about_ca_system_score_gemma":0.0010621945,"threshold_uncertainty_score":0.007974923},"labels":[],"label_agreement":null},{"id":"W3040858112","doi":"10.1002/stvr.1745","title":"TimelyRep: Timing deterministic replay for Android web applications","year":2020,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Green IT and Sustainability","field":"Engineering","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Debugging; Computer science; Android (operating system); Web application; Touchscreen; Backward compatibility; Mobile device; Operating system; Embedded system; Event (particle physics); World Wide Web","score_opus":0.030880898994668088,"score_gpt":0.24372867710065063,"score_spread":0.21284777810598254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3040858112","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3377876,0.0024057426,0.36452144,0.00045639477,0.0004976241,0.0008072257,0.0016228011,0.28342098,0.008480191],"genre_scores_gemma":[0.931396,0.000288439,0.058711864,0.00016263596,0.00005574272,0.00026754162,0.0012284106,0.0036132794,0.004276124],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99804366,0.00033412434,0.00023140365,0.00031731068,0.0008435102,0.0002299831],"domain_scores_gemma":[0.99462575,0.0016942498,0.0006265835,0.001586877,0.0012516024,0.00021482806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014425995,0.0014667761,0.0005227049,0.0013855301,0.000368966,0.0007002935,0.001832431,0.000551314,0.003140394],"category_scores_gemma":[0.0072545586,0.00056890765,0.00053556764,0.00032919474,0.00039929437,0.0010415169,0.0009711254,0.0009175429,0.001041698],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0070067667,0.0006929637,0.02645311,0.0020676428,0.00042273378,0.0032353401,0.0016765237,0.07167737,0.20157695,0.0077405246,0.05299282,0.62445724],"study_design_scores_gemma":[0.0006007059,0.0025720447,0.015528948,0.00014851487,0.00029712508,0.0016958788,0.00022564066,0.6900192,0.25078267,0.002184983,0.03560224,0.00034211608],"about_ca_topic_score_codex":0.004064432,"about_ca_topic_score_gemma":0.0031219816,"teacher_disagreement_score":0.004064432,"about_ca_system_score_codex":0.0004708933,"about_ca_system_score_gemma":0.0009043322,"threshold_uncertainty_score":0.010505676},"labels":[],"label_agreement":null},{"id":"W3092453472","doi":"10.1002/stvr.1751","title":"BUGSJS: a benchmark and taxonomy of JavaScript bugs","year":2020,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"European Social Fund; European Commission; Natural Sciences and Engineering Research Council of Canada; Advanced Remanufacturing and Technology Centre; National Research, Development and Innovation Office; Innovációs és Technológiai Minisztérium","keywords":"JavaScript; Computer science; Unobtrusive JavaScript; Benchmark (surveying); Unit testing; Software bug; Debugging; Taxonomy (biology); Programming language; Web application; Software; Software testing; Test case; Software engineering; Rich Internet application; World Wide Web; Machine learning","score_opus":0.05055143313862166,"score_gpt":0.23784943200647207,"score_spread":0.18729799886785042,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3092453472","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.836064,0.0066206027,0.09647378,0.0011950716,0.00039056546,0.0013581854,0.022664804,0.027221845,0.0080111325],"genre_scores_gemma":[0.78612214,0.0016117098,0.14813918,0.00036205744,0.00011198211,0.0013654436,0.056822624,0.0033294882,0.0021353466],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9837818,0.003243753,0.002657262,0.0018715583,0.0075621326,0.0008834231],"domain_scores_gemma":[0.9476385,0.023495547,0.007570923,0.0058816806,0.013668538,0.0017447505],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007724821,0.0014590021,0.0006202083,0.009900055,0.0010023968,0.0016821609,0.002259229,0.0011249069,0.000761656],"category_scores_gemma":[0.03974611,0.0005094592,0.0009711666,0.005282644,0.0011550979,0.0019736437,0.0022793303,0.001240978,0.0004992518],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015065097,0.0018618307,0.3265269,0.010159151,0.0006476183,0.0020378362,0.0050823567,0.07144265,0.046945777,0.0148598645,0.07948637,0.43944308],"study_design_scores_gemma":[0.0004988544,0.0037590258,0.31582505,0.002588079,0.00053989014,0.0046198107,0.003468318,0.4382162,0.08348332,0.02421858,0.12219081,0.0005920086],"about_ca_topic_score_codex":0.006062353,"about_ca_topic_score_gemma":0.007132469,"teacher_disagreement_score":0.009900055,"about_ca_system_score_codex":0.0013200011,"about_ca_system_score_gemma":0.0022717773,"threshold_uncertainty_score":0.040853262},"labels":[],"label_agreement":null},{"id":"W3206885924","doi":"10.1002/stvr.1799","title":"A mutation framework for evaluating security analysis tools in IoT applications","year":2021,"lang":"en","type":"preprint","venue":"Software Testing Verification and Reliability","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Taint checking; Internet of Things; Sensitivity (control systems); Precision and recall; Context (archaeology); Security analysis; Set (abstract data type); Data mining; Process (computing); Domain (mathematical analysis); Information flow; Information retrieval; Computer security; Software; Engineering; Programming language","score_opus":0.06271620279564884,"score_gpt":0.3700882404693309,"score_spread":0.3073720376736821,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3206885924","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15326987,0.00029155132,0.8311108,0.00027277457,0.000039316135,0.0007759287,0.0004128654,0.01125052,0.00257638],"genre_scores_gemma":[0.5117413,0.000080932274,0.48608157,0.000112086884,0.000014813505,0.00046990742,0.0005562343,0.00037624408,0.00056698907],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98828447,0.004022726,0.001077988,0.0011141263,0.004991261,0.0005094658],"domain_scores_gemma":[0.98207134,0.009867064,0.0022115281,0.0020163076,0.0035023892,0.00033138867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008876107,0.0018832827,0.0009020873,0.0063809277,0.0007876697,0.0018152567,0.0018381526,0.0013208657,0.00090930634],"category_scores_gemma":[0.022615451,0.0004958877,0.0018588621,0.0012256672,0.0020586995,0.0018360645,0.0017559254,0.0013644912,0.0002457715],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007519165,0.0014226434,0.03192598,0.0007415953,0.00047470222,0.00060674764,0.0008363883,0.4907491,0.12563528,0.06325949,0.004166594,0.2794295],"study_design_scores_gemma":[0.000057235007,0.0003453214,0.0029678212,0.00009070297,0.0000774991,0.00021372408,0.000090681,0.9467666,0.03568222,0.011313062,0.0023301726,0.00006491752],"about_ca_topic_score_codex":0.0074922405,"about_ca_topic_score_gemma":0.005052852,"teacher_disagreement_score":0.008876107,"about_ca_system_score_codex":0.0025349269,"about_ca_system_score_gemma":0.0031176421,"threshold_uncertainty_score":0.046941876},"labels":[],"label_agreement":null},{"id":"W3207636857","doi":"10.1002/stvr.1796","title":"GPU acceleration of finite state machine input execution: Improving scale and performance","year":2021,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Engineering and Physical Sciences Research Council; University of Edinburgh","keywords":"Computer science; Scalability; Parallel computing; Kernel (algebra); Speedup; Multi-core processor; Finite-state machine; Acceleration; Process (computing); CUDA; Computer engineering; Algorithm; Programming language; Operating system","score_opus":0.02022802077216418,"score_gpt":0.24044934009781616,"score_spread":0.22022131932565198,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3207636857","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.67771345,0.0029282297,0.23497503,0.00082568213,0.0006153285,0.0002513567,0.0010310018,0.055176537,0.02648347],"genre_scores_gemma":[0.85453975,0.00031806002,0.13994682,0.00013407612,0.000030772313,0.000108715125,0.0013245785,0.0011964111,0.0024008465],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99875724,0.00024306963,0.00007334766,0.00023171733,0.0005072189,0.00018745742],"domain_scores_gemma":[0.99684864,0.0013139572,0.00013133226,0.0008716639,0.00069740094,0.00013699493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010774153,0.00102729,0.00068593374,0.00084503525,0.0003760081,0.0010672203,0.0020596255,0.000504174,0.004973635],"category_scores_gemma":[0.0054893843,0.00043343086,0.0007423137,0.0010173973,0.00038692376,0.0013636592,0.0008476615,0.00111137,0.0011970497],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031042078,0.00092328625,0.024845457,0.0009733742,0.00037813364,0.00048697423,0.00065380044,0.24930945,0.117883526,0.011628344,0.036509357,0.5533042],"study_design_scores_gemma":[0.00013156854,0.00025615236,0.0027711536,0.000036918333,0.00005388137,0.000057297595,0.00006593296,0.94789326,0.039926264,0.002085867,0.006690685,0.000030939136],"about_ca_topic_score_codex":0.008710368,"about_ca_topic_score_gemma":0.009333426,"teacher_disagreement_score":0.008710368,"about_ca_system_score_codex":0.0011081463,"about_ca_system_score_gemma":0.0011303115,"threshold_uncertainty_score":0.017319322},"labels":[],"label_agreement":null},{"id":"W4230255398","doi":"10.1002/stvr.385","title":"Error‐preserving reductions on communication protocols","year":2007,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Reachability; Computer science; Model checking; Variety (cybernetics); State (computer science); Process (computing); Transformation (genetics); Formal methods; Communications protocol; Distributed computing; Theoretical computer science; Programming language; Artificial intelligence; Computer network","score_opus":0.08466286065561494,"score_gpt":0.36484616857315927,"score_spread":0.2801833079175443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4230255398","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.103151485,0.00015172469,0.87477523,0.0007059572,0.00013713536,0.00035867142,0.00014920668,0.004747424,0.015823152],"genre_scores_gemma":[0.77367646,0.00030027123,0.2171172,0.00024659545,0.00008064595,0.0006810066,0.00034081584,0.00072949287,0.006827435],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9934277,0.002158262,0.0004116382,0.00046216097,0.0029620545,0.00057819294],"domain_scores_gemma":[0.9815481,0.011083377,0.0010974578,0.004426183,0.0016076488,0.0002374149],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030459636,0.0009936326,0.0007373836,0.0011322518,0.0007407653,0.0011669714,0.0016536643,0.00092617614,0.0033184197],"category_scores_gemma":[0.01623099,0.000488622,0.0014008464,0.00071705563,0.0025028118,0.0018571307,0.0029505333,0.002581761,0.0007991883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000611738,0.0004441231,0.0009784411,0.00061252527,0.00011077378,0.0017529046,0.0012771626,0.2509223,0.055313375,0.5452724,0.005173283,0.13753088],"study_design_scores_gemma":[0.00016868876,0.00021878927,0.00042009001,0.00006960623,0.00011197385,0.0005524702,0.0002607739,0.33190644,0.1258149,0.5207656,0.019658655,0.000052083353],"about_ca_topic_score_codex":0.000738227,"about_ca_topic_score_gemma":0.00057592103,"teacher_disagreement_score":0.0033184197,"about_ca_system_score_codex":0.00067240925,"about_ca_system_score_gemma":0.0012023315,"threshold_uncertainty_score":0.016108751},"labels":[],"label_agreement":null},{"id":"W4236669998","doi":"10.1002/stvr.401","title":"Modelling methods for web application verification and testing: state of the art","year":2008,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Web testing; Software engineering; Software; Web application; Verification and validation; Web modeling; Software testing; Data mining; World Wide Web; Web service; Programming language; Engineering","score_opus":0.06046944475577618,"score_gpt":0.3077181642846764,"score_spread":0.24724871952890023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4236669998","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022149896,0.029323732,0.9630346,0.00042306102,0.00008295322,0.00008165259,0.000058156747,0.0008743589,0.0039064987],"genre_scores_gemma":[0.13574481,0.05162431,0.80689394,0.00032033774,0.0004370692,0.00046643146,0.00055336254,0.00059134274,0.0033684538],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98904306,0.004453625,0.000701218,0.0008807558,0.00470952,0.00021190345],"domain_scores_gemma":[0.9779936,0.016656462,0.0011332243,0.0023761117,0.0016784834,0.00016210652],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0070502628,0.0018626117,0.0021252148,0.0053510773,0.0005704704,0.0042960313,0.003863483,0.0026575564,0.0043265405],"category_scores_gemma":[0.017430214,0.0013564231,0.0025010079,0.005152005,0.0032176848,0.005006043,0.002005405,0.0025396757,0.002087678],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000102984966,0.0001964529,0.0016082665,0.0035997499,0.00031214353,0.00021212717,0.00042255115,0.07489446,0.0041866307,0.22736327,0.0044045327,0.68269676],"study_design_scores_gemma":[0.000074735144,0.000092941205,0.000831822,0.001986859,0.00013723249,0.0004824691,0.0001533318,0.5906611,0.005176758,0.31984955,0.080430984,0.00012218965],"about_ca_topic_score_codex":0.0028793486,"about_ca_topic_score_gemma":0.0010959003,"teacher_disagreement_score":0.9929497,"about_ca_system_score_codex":0.0013370902,"about_ca_system_score_gemma":0.0012345415,"threshold_uncertainty_score":0.037285805},"labels":[],"label_agreement":null},{"id":"W4248647026","doi":"10.1002/stvr.410","title":"Improving the coverage criteria of UML state machines using data flow analysis","year":2009,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Unified Modeling Language; Guard (computer science); Finite-state machine; Data mining; Data flow diagram; Test suite; Context (archaeology); State (computer science); Control flow; Data-flow analysis; Tree (set theory); Test case; Algorithm; Machine learning; Database; Programming language; Mathematics; Software","score_opus":0.04962212616320409,"score_gpt":0.3113156837427992,"score_spread":0.2616935575795951,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248647026","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34594256,0.00028792542,0.64980954,0.00026482882,0.000009750065,0.00022709157,0.00026866587,0.0015208675,0.0016687914],"genre_scores_gemma":[0.84857213,0.00008268785,0.15042144,0.000038452876,0.000011474681,0.00019696762,0.00037591715,0.00009891789,0.00020196383],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.99191767,0.004424823,0.00046736933,0.0004991555,0.0022930868,0.00039778653],"domain_scores_gemma":[0.9326225,0.0572468,0.0034577195,0.001940481,0.004344654,0.00038782996],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067224316,0.0010055059,0.00086887006,0.0067187496,0.000487898,0.0015717188,0.00087843405,0.00090583577,0.00093094737],"category_scores_gemma":[0.041230783,0.00046783884,0.001149456,0.0015596242,0.0011434419,0.0021754527,0.0013211973,0.0006263652,0.00013116981],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008056375,0.0004528644,0.048314147,0.0005080661,0.00018972535,0.00061050785,0.0010754016,0.5931373,0.03975538,0.030341052,0.001195753,0.2836141],"study_design_scores_gemma":[0.000031302025,0.00014085112,0.0025002747,0.000048215752,0.000035640533,0.00007986661,0.00006091521,0.9727895,0.015765525,0.007932138,0.0005965186,0.00001927684],"about_ca_topic_score_codex":0.0042762244,"about_ca_topic_score_gemma":0.0027272059,"teacher_disagreement_score":0.0067224316,"about_ca_system_score_codex":0.0016750196,"about_ca_system_score_gemma":0.001255769,"threshold_uncertainty_score":0.035552084},"labels":[],"label_agreement":null},{"id":"W4251245562","doi":"10.1002/stvr.396","title":"Transition covering tests for systems with queues","year":2008,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal; Nortel (Canada)","funders":"","keywords":"Context (archaeology); Computer science; Queue; Metric (unit); Concurrency; Cover (algebra); Model-based testing; Test (biology); Test case; Reliability engineering; Distributed computing; Programming language; Engineering; Operations management; Mechanical engineering","score_opus":0.0432607937311662,"score_gpt":0.25965007470477164,"score_spread":0.21638928097360544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251245562","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10449646,0.00022594816,0.890035,0.00015591813,0.000040206174,0.0001018095,0.00008203561,0.0020711687,0.0027915246],"genre_scores_gemma":[0.82992953,0.00012014106,0.16847193,0.00008785753,0.000043742213,0.00013157444,0.00018648585,0.00019845032,0.0008304048],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9960245,0.0013201302,0.0002981786,0.0005036467,0.0014506435,0.00040293924],"domain_scores_gemma":[0.9833786,0.013357203,0.0008699974,0.0010904087,0.0009893054,0.00031453677],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018609673,0.00060277455,0.0007121117,0.0014664794,0.0005490912,0.001397303,0.00094878714,0.0009506403,0.0017572214],"category_scores_gemma":[0.013760338,0.00039174236,0.0010133635,0.0008416457,0.0018284294,0.002040806,0.0019134635,0.0011909498,0.00021952765],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012134722,0.00029851368,0.01318149,0.00062985695,0.0002096593,0.0037373581,0.0023626315,0.35861984,0.069423415,0.29398885,0.0025851484,0.25374973],"study_design_scores_gemma":[0.00006892056,0.00029037485,0.0016723829,0.00012695136,0.000067151275,0.00093020767,0.00015509583,0.81108135,0.05254556,0.1258835,0.007112031,0.00006647371],"about_ca_topic_score_codex":0.0015122182,"about_ca_topic_score_gemma":0.0007698928,"teacher_disagreement_score":0.0018609673,"about_ca_system_score_codex":0.00072306825,"about_ca_system_score_gemma":0.0008125063,"threshold_uncertainty_score":0.0098418},"labels":[],"label_agreement":null},{"id":"W4322489317","doi":"10.1002/stvr.1842","title":"An investigation of distributed computing for combinatorial testing","year":2023,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hypergraph; Computer science; Graph; Theoretical computer science; Vertex (graph theory); Algorithm; Parallel computing; Mathematics; Discrete mathematics","score_opus":0.0456312420782707,"score_gpt":0.2955049233390974,"score_spread":0.2498736812608267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4322489317","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12951382,0.0007234374,0.8289721,0.0026559946,0.00018012563,0.00017564875,0.000052254494,0.0012969696,0.036429748],"genre_scores_gemma":[0.81050545,0.00022717204,0.18516716,0.00020965365,0.00006151509,0.0001357649,0.00005544117,0.00017420294,0.0034636545],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99726164,0.0012834831,0.00007884083,0.00041636798,0.0006438791,0.00031577653],"domain_scores_gemma":[0.9887156,0.007150509,0.00038894126,0.0023477983,0.0010440721,0.00035309844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024632916,0.00049679633,0.0006236676,0.0007277483,0.00086924405,0.0020726672,0.0018419991,0.00078613026,0.0045548202],"category_scores_gemma":[0.008252601,0.0003066521,0.00070264266,0.0010873507,0.0019665277,0.0025683895,0.0015724255,0.0017156082,0.00045314708],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037416277,0.0003156014,0.0035396584,0.00017920393,0.00007288937,0.00041123494,0.00034032736,0.3696015,0.012492608,0.43137625,0.0040332074,0.17726345],"study_design_scores_gemma":[0.00005948031,0.00011484345,0.00043104732,0.00002646798,0.000024888113,0.00013967913,0.00009985949,0.8714729,0.004807635,0.117901094,0.0049089883,0.0000131613615],"about_ca_topic_score_codex":0.0033074613,"about_ca_topic_score_gemma":0.0027534072,"teacher_disagreement_score":0.0045548202,"about_ca_system_score_codex":0.0022579837,"about_ca_system_score_gemma":0.0014880402,"threshold_uncertainty_score":0.016382933},"labels":[],"label_agreement":null}]}