{"meta":{"query_hash":"100aa09bc2bb","filters":{"topic":"Software Testing and Debugging Techniques"},"cohort_total":996,"direct_labels_cover":2,"predictions_cover":996,"exported":996,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/100aa09bc2bb","api":"https://metacan.xera.ac/api/v1/cohort?topic=Software+Testing+and+Debugging+Techniques"},"results":[{"id":"W110592620","doi":"10.5753/sbes.2007.21316","title":"Experimental Evaluation of Coverage Criteria for FSM-based Testing","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Fundação de Amparo à Pesquisa do Estado de São Paulo","keywords":"Computer science; Code coverage; Finite-state machine; Reliability engineering; Test plan; Test strategy; Empirical research; Test (biology); Compromise; Plan (archaeology); Data mining; Machine learning; Artificial intelligence; Algorithm; Engineering; Mathematics; Statistics; Software","score_opus":0.1379830994562716,"score_gpt":0.4016068810919319,"score_spread":0.26362378163566025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W110592620","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9508592,0.00082304195,0.043713264,0.00015337695,0.00004892036,0.0005034949,0.0006328227,0.001082039,0.0021838692],"genre_scores_gemma":[0.9555466,0.00020735429,0.04208954,0.000036174286,0.000018960054,0.00045292935,0.00090748514,0.0001759731,0.0005650058],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98655164,0.006220367,0.0013728747,0.0011578883,0.0041595264,0.00053777784],"domain_scores_gemma":[0.83051264,0.14627168,0.006153734,0.0071554445,0.008656008,0.0012504555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009592774,0.0013345382,0.00081229734,0.002263372,0.0005371223,0.00064583233,0.0018627741,0.0016409579,0.0023505029],"category_scores_gemma":[0.071484335,0.0004047756,0.00042397872,0.0016598358,0.001256949,0.0014564922,0.0011647757,0.00076243887,0.0002924181],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013616712,0.011818793,0.020171981,0.003584436,0.00060494477,0.00055822224,0.0015016821,0.37601522,0.2847145,0.0065368284,0.0033377511,0.2775389],"study_design_scores_gemma":[0.0012945209,0.016885564,0.019359607,0.00016454008,0.0002726923,0.00049414893,0.0003882494,0.69610894,0.2590599,0.0031973766,0.0026483925,0.00012602725],"about_ca_topic_score_codex":0.0020412218,"about_ca_topic_score_gemma":0.002015279,"teacher_disagreement_score":0.009592774,"about_ca_system_score_codex":0.0013779793,"about_ca_system_score_gemma":0.00057421584,"threshold_uncertainty_score":0.050732017},"labels":[],"label_agreement":null},{"id":"W112845331","doi":"10.1007/978-3-662-45501-2_19","title":"Impact Analysis via Reachability and Alias Analysis","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in business information processing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Alias; Reachability; Computer science; False positive paradox; Aliasing; Test suite; Static analysis; Pointer analysis; Suite; Pointer (user interface); Set (abstract data type); Program analysis; False positives and false negatives; Path (computing); Continuation; Test case; Theoretical computer science; Data mining; Programming language; Machine learning; Artificial intelligence","score_opus":0.011863693441486176,"score_gpt":0.2617542064627766,"score_spread":0.2498905130212904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W112845331","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009062239,0.0004967376,0.97736996,0.00021101526,0.000032842792,0.00015140773,0.00041295125,0.0037853054,0.008477533],"genre_scores_gemma":[0.31598786,0.0013176227,0.6717746,0.00024063331,0.0001184006,0.00047530144,0.001817182,0.0014523882,0.00681599],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99615246,0.0008815826,0.00017384291,0.00047959713,0.0020017682,0.00031069774],"domain_scores_gemma":[0.99230313,0.00537917,0.0006330237,0.00073954207,0.00085901114,0.00008620011],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002294641,0.0021804431,0.0010483,0.009997662,0.00092773227,0.0024125825,0.0020679887,0.0011697046,0.0083640795],"category_scores_gemma":[0.010734707,0.00080914417,0.0031098418,0.004165605,0.001816709,0.004134559,0.0026151736,0.0022244935,0.0016740966],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022030655,0.00023640222,0.01017532,0.0007572666,0.00032665255,0.0008658531,0.0004731504,0.29127136,0.018976025,0.18187045,0.0064692074,0.48835802],"study_design_scores_gemma":[0.00001744961,0.00010699873,0.003268963,0.00020478938,0.00020155578,0.00053206336,0.00015758797,0.7069434,0.015466366,0.262519,0.010484536,0.00009731545],"about_ca_topic_score_codex":0.005254914,"about_ca_topic_score_gemma":0.004810606,"teacher_disagreement_score":0.009997662,"about_ca_system_score_codex":0.0020666036,"about_ca_system_score_gemma":0.0014822114,"threshold_uncertainty_score":0.027980626},"labels":[],"label_agreement":null},{"id":"W1151207335","doi":"","title":"Transformation by example","year":2010,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Transformation (genetics); Computer science; Set (abstract data type); Context (archaeology); Source code; Function (biology); Process (computing); Algorithm; Theoretical computer science; Artificial intelligence; Programming language","score_opus":0.01585661375620726,"score_gpt":0.26565639809918845,"score_spread":0.2497997843429812,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1151207335","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008171328,0.00071359193,0.7506909,0.0022231862,0.000931981,0.001095121,0.009620509,0.041940648,0.1846127],"genre_scores_gemma":[0.1624474,0.0017001169,0.6556527,0.0014769223,0.00030906132,0.0015787347,0.02657774,0.011860096,0.1383973],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9978478,0.00045412628,0.00019421028,0.00048298453,0.00081628293,0.00020460371],"domain_scores_gemma":[0.9976236,0.00068514573,0.00008548242,0.0011699455,0.0003700948,0.00006569284],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011550615,0.0012065351,0.000637436,0.0014548787,0.0007645491,0.0031541518,0.001874131,0.0012235314,0.09217904],"category_scores_gemma":[0.0063480223,0.0005047534,0.0020265887,0.0013736914,0.00092113874,0.004437492,0.0039503174,0.001953531,0.031783786],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028368598,0.00019715224,0.0015855897,0.0008859064,0.00008961624,0.00081165496,0.00061889127,0.012827728,0.007032466,0.3734355,0.19778055,0.40445134],"study_design_scores_gemma":[0.00005587765,0.00007042436,0.00036305276,0.00014140087,0.000037566857,0.0007434692,0.00016064235,0.029477037,0.00761753,0.12563546,0.8356588,0.000038745242],"about_ca_topic_score_codex":0.0021110715,"about_ca_topic_score_gemma":0.0023061365,"teacher_disagreement_score":0.09217904,"about_ca_system_score_codex":0.0009845622,"about_ca_system_score_gemma":0.0012158672,"threshold_uncertainty_score":0.30836958},"labels":[],"label_agreement":null},{"id":"W118100860","doi":"","title":"Test Generation Based On Control And Data Dependencies Within Multi-Process SDL Specifications.","year":2000,"lang":"en","type":"article","venue":"Publikationsdatenbank der Fraunhofer-Gesellschaft (Fraunhofer-Gesellschaft)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Interleaving; Control flow; Process (computing); Data flow diagram; Test data; Data modeling; Data-flow analysis; Distributed computing; Programming language; Database; Operating system","score_opus":0.10154280709632142,"score_gpt":0.30717166234079046,"score_spread":0.20562885524446906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W118100860","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04655245,0.00025882616,0.93929374,0.00036536335,0.00009507147,0.00035422092,0.0005432161,0.0074623697,0.005074711],"genre_scores_gemma":[0.65347755,0.00018590166,0.33851665,0.00024984928,0.000030760795,0.000323039,0.001862034,0.0011182667,0.0042358404],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99642855,0.0015117887,0.0002625109,0.0003929253,0.001164288,0.00023994796],"domain_scores_gemma":[0.9843115,0.011932145,0.00070874096,0.0014696012,0.0013792124,0.0001988303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002922566,0.0008072755,0.00039384898,0.0012546184,0.00042612886,0.0011267795,0.0010636359,0.000972986,0.0059725964],"category_scores_gemma":[0.016156586,0.000516212,0.0008083177,0.00049950194,0.0010241483,0.001789113,0.0011710058,0.0010462252,0.00088269374],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002599305,0.00062610995,0.009740355,0.0010024826,0.00018140595,0.002361763,0.0010192731,0.2109141,0.09017701,0.084011726,0.015562371,0.5818041],"study_design_scores_gemma":[0.0002328272,0.00041522292,0.0012102387,0.00019444092,0.00009038145,0.0006943755,0.00010597855,0.8083513,0.1330532,0.044433184,0.011169873,0.00004893228],"about_ca_topic_score_codex":0.0021498697,"about_ca_topic_score_gemma":0.002887436,"teacher_disagreement_score":0.0059725964,"about_ca_system_score_codex":0.0007857741,"about_ca_system_score_gemma":0.0012084461,"threshold_uncertainty_score":0.019980311},"labels":[],"label_agreement":null},{"id":"W120366370","doi":"","title":"Towards the Verification and Validation of DEVS Models","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"DEVS; Rotation formalisms in three dimensions; Computer science; Software engineering; Software; Verification and validation of computer simulation models; Context (archaeology); Modeling and simulation; Programming language; Simulation; Mathematics","score_opus":0.0534625555270374,"score_gpt":0.27175700200499026,"score_spread":0.21829444647795285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W120366370","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010386111,0.00020831487,0.9838111,0.0006230169,0.00008339499,0.000101501806,0.00011856142,0.0014325767,0.0032354502],"genre_scores_gemma":[0.19416003,0.00085239526,0.79926485,0.00047606605,0.00008871876,0.0003303868,0.0008993898,0.0005952423,0.0033329846],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98618966,0.0048853247,0.00093018275,0.0008866065,0.0065463795,0.00056190504],"domain_scores_gemma":[0.9697851,0.011164184,0.0017844398,0.008564245,0.008205156,0.0004969441],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0136197535,0.0008929818,0.0007235583,0.0018878747,0.0008915153,0.0034971584,0.00294918,0.0021339415,0.0023751773],"category_scores_gemma":[0.041730125,0.0011219815,0.0017489434,0.0008440828,0.003956732,0.0041161054,0.0038535877,0.0037915583,0.0009771897],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019492082,0.00017139019,0.004571242,0.0008080777,0.00010828057,0.00092856557,0.0013884095,0.20789503,0.02395086,0.65636677,0.004628235,0.09898831],"study_design_scores_gemma":[0.000088336354,0.00013568056,0.00055498775,0.000529043,0.00007167067,0.00042773562,0.00025096023,0.6495973,0.055714708,0.22151822,0.071042165,0.000069166206],"about_ca_topic_score_codex":0.0045209127,"about_ca_topic_score_gemma":0.0031948602,"teacher_disagreement_score":0.0136197535,"about_ca_system_score_codex":0.0017921566,"about_ca_system_score_gemma":0.0050312057,"threshold_uncertainty_score":0.072028995},"labels":[],"label_agreement":null},{"id":"W125925145","doi":"10.1007/978-3-319-04921-2_15","title":"Solving Equations on Words with Morphisms and Antimorphisms","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Solver; Boolean satisfiability problem; Computer science; Range (aeronautics); Morphism; Theoretical computer science; Representation (politics); Graph; System of linear equations; Satisfiability; Boolean data type; Algorithm; Applied mathematics; Discrete mathematics; Mathematics; Programming language","score_opus":0.020719302508557077,"score_gpt":0.24294149421585456,"score_spread":0.22222219170729748,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W125925145","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034004815,0.00045574328,0.94015056,0.0006841583,0.00021188232,0.00007378776,0.00017170308,0.0006456117,0.023601731],"genre_scores_gemma":[0.26301506,0.0010265375,0.7035009,0.00034748417,0.00023340523,0.00014561605,0.00061118614,0.00080513855,0.030314613],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9986237,0.0004213297,0.00017104151,0.00028741203,0.00035356803,0.00014296669],"domain_scores_gemma":[0.9984609,0.0010966358,0.00006753874,0.00016456151,0.0001724131,0.00003795415],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011736143,0.0011037745,0.0010795848,0.00084357586,0.0008123577,0.0021641776,0.0015400551,0.0011980844,0.011202545],"category_scores_gemma":[0.0058593918,0.00092790165,0.0022517198,0.001416686,0.0018798296,0.006697804,0.0038724283,0.0029312451,0.0021956165],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000390135,0.000036668393,0.00051110703,0.00039805664,0.000049306207,0.00017807112,0.0009587385,0.008435242,0.0026963057,0.88347876,0.0048924326,0.098326266],"study_design_scores_gemma":[0.000021770273,0.000013462691,0.00007306214,0.000045710873,0.000030125173,0.00009838789,0.00013508792,0.016860709,0.002237314,0.96968377,0.01078392,0.000016790804],"about_ca_topic_score_codex":0.0009837112,"about_ca_topic_score_gemma":0.0014209545,"teacher_disagreement_score":0.011202545,"about_ca_system_score_codex":0.0007396903,"about_ca_system_score_gemma":0.00070503895,"threshold_uncertainty_score":0.03747624},"labels":[],"label_agreement":null},{"id":"W127597465","doi":"10.1007/978-3-642-39955-8_7","title":"Automatically Repairing Concurrency Bugs with ARC","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Concurrency; Crossover; Correctness; Java; Synchronization (alternating current); Programming language; Genetic algorithm; Parallel computing; Distributed computing; Artificial intelligence","score_opus":0.01690706949741025,"score_gpt":0.2442023236193368,"score_spread":0.22729525412192655,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W127597465","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037413035,0.0006083513,0.9180197,0.00026340946,0.00016577465,0.00009945241,0.00028741418,0.038244855,0.0048981146],"genre_scores_gemma":[0.26946038,0.00029687057,0.7202854,0.000112953174,0.000039201062,0.0000568039,0.0008045761,0.0027127333,0.0062310915],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99824166,0.00025635873,0.00013130448,0.0004198341,0.0007944188,0.00015640259],"domain_scores_gemma":[0.9933009,0.00325048,0.0005490869,0.0020063054,0.00077792205,0.000115238305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012346884,0.0014207541,0.0010475212,0.0019701885,0.0005811415,0.00095559214,0.002663564,0.0009975629,0.008657321],"category_scores_gemma":[0.00620644,0.0009516924,0.0011498005,0.0015168786,0.0012236492,0.0030550852,0.0022340552,0.0016462557,0.0019354664],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005457709,0.00026628928,0.0033555657,0.0011093545,0.00011144014,0.0006525152,0.00038849088,0.06316775,0.026838077,0.019808112,0.012730523,0.87102616],"study_design_scores_gemma":[0.0002210534,0.0003751219,0.0014028528,0.00024042855,0.00035413753,0.002087845,0.00026665756,0.75633514,0.13245559,0.072130136,0.034015983,0.00011495397],"about_ca_topic_score_codex":0.0020851276,"about_ca_topic_score_gemma":0.0028479276,"teacher_disagreement_score":0.008657321,"about_ca_system_score_codex":0.00045937972,"about_ca_system_score_gemma":0.0010667004,"threshold_uncertainty_score":0.028961599},"labels":[],"label_agreement":null},{"id":"W1416149289","doi":"10.1007/978-3-642-29860-8_16","title":"Efficient Techniques for Near-Optimal Instrumentation in Time-Triggered Runtime Verification","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Heuristics; Overhead (engineering); Runtime verification; Instrumentation (computer programming); Set (abstract data type); Execution time; State (computer science); Completeness (order theory); Parallel computing; Real-time computing; Distributed computing; Formal verification; Algorithm; Programming language; Operating system","score_opus":0.016948309923076294,"score_gpt":0.26222442469566104,"score_spread":0.24527611477258474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1416149289","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030506053,0.00026352427,0.99311674,0.00007765115,0.000050415983,0.000039296858,0.000033607575,0.0019372785,0.0014309145],"genre_scores_gemma":[0.24182749,0.0005105751,0.7523214,0.00020046273,0.00010158502,0.00019111623,0.00025420947,0.0010385241,0.0035546294],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9952453,0.0011288003,0.00037948904,0.00071416545,0.0018897505,0.0006424174],"domain_scores_gemma":[0.99193734,0.004694059,0.00041112243,0.0022212418,0.0006029203,0.0001332269],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020965522,0.0018895305,0.0017398926,0.0014348058,0.0010013536,0.0018089234,0.0031230873,0.0014119605,0.0068722996],"category_scores_gemma":[0.010214452,0.0015086052,0.0020419087,0.0016138193,0.002125389,0.0041832547,0.00419677,0.004123205,0.0021560376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011191854,0.00024643078,0.000839664,0.0008703299,0.00017307728,0.00032686742,0.00053566473,0.12457846,0.06970077,0.19851771,0.007423186,0.5956686],"study_design_scores_gemma":[0.00013916058,0.00020191804,0.0003983128,0.0001341375,0.00016844439,0.00038185698,0.00010950667,0.6464673,0.051429212,0.29101142,0.009476949,0.000081817656],"about_ca_topic_score_codex":0.0011506836,"about_ca_topic_score_gemma":0.0026389596,"teacher_disagreement_score":0.0068722996,"about_ca_system_score_codex":0.0012247214,"about_ca_system_score_gemma":0.0019103551,"threshold_uncertainty_score":0.022990108},"labels":[],"label_agreement":null},{"id":"W1471943874","doi":"10.1007/978-3-642-38230-7_7","title":"SiteHopper: Abstracting Navigation State Machines for the Efficient Verification of Web Applications","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Computer science; Web crawler; Abstraction; State (computer science); Model checking; Finite-state machine; On the fly; Data mining; Programming language; Information retrieval; World Wide Web; Operating system","score_opus":0.01938348445182148,"score_gpt":0.2701109497937844,"score_spread":0.25072746534196294,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1471943874","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005437366,0.00010572811,0.98041934,0.00005013646,0.000045498247,0.000057041478,0.00020815595,0.012450334,0.0012264869],"genre_scores_gemma":[0.24552114,0.00037585478,0.73998916,0.00015295293,0.00006338885,0.00030817208,0.0014906684,0.0038230922,0.008275585],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988657,0.00031120342,0.00007619897,0.00021483983,0.00043087557,0.000101083264],"domain_scores_gemma":[0.9985917,0.00075826497,0.000079099795,0.0003835641,0.00016144963,0.000025936679],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010276042,0.0015190598,0.00096325733,0.0009332876,0.00055696623,0.0012553446,0.0021841251,0.0011910615,0.01135878],"category_scores_gemma":[0.0036528006,0.0012545175,0.0016119797,0.0006843748,0.0013264328,0.0034875881,0.0020470512,0.0021545845,0.0025774427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005976338,0.00019782648,0.0018183459,0.0012536969,0.00018648524,0.000445079,0.000470003,0.25148562,0.047276907,0.16606708,0.024785198,0.5054161],"study_design_scores_gemma":[0.00006860809,0.000087323664,0.00039920077,0.00009099225,0.00006197994,0.0001601327,0.00003103352,0.80089164,0.04230149,0.13845253,0.017408263,0.000046801557],"about_ca_topic_score_codex":0.0032521945,"about_ca_topic_score_gemma":0.005145703,"teacher_disagreement_score":0.01135878,"about_ca_system_score_codex":0.00064364297,"about_ca_system_score_gemma":0.0012645057,"threshold_uncertainty_score":0.037998855},"labels":[],"label_agreement":null},{"id":"W1479736339","doi":"10.1007/11813040_27","title":"Towards Automatic Exception Safety Verification","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Programming language; Exception handling; Java; Static analysis; Raising (metalworking); Software engineering; Software","score_opus":0.018498283578180015,"score_gpt":0.2547942841712107,"score_spread":0.2362960005930307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1479736339","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037422339,0.00027641925,0.9846348,0.00017614417,0.00013815658,0.00005552325,0.000043271786,0.0047133146,0.006220104],"genre_scores_gemma":[0.17882755,0.0006557046,0.8016768,0.0004573201,0.00018881897,0.00012437873,0.00053166196,0.001439927,0.016097877],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996786,0.0005725443,0.00019198091,0.00047419057,0.0017302389,0.00024507707],"domain_scores_gemma":[0.99500895,0.0020505544,0.00018441664,0.0018621149,0.0008275691,0.000066352324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018706449,0.0011795335,0.0010640251,0.0018609398,0.0009467446,0.00211886,0.0021264965,0.0013115722,0.008948051],"category_scores_gemma":[0.006084174,0.0014787898,0.0016634151,0.00097207143,0.0017743068,0.0036339127,0.0037087696,0.00334069,0.0042362213],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025904534,0.00016570666,0.0006874133,0.00055606215,0.000087702436,0.00035575387,0.00038244773,0.021168174,0.04736801,0.21429487,0.016357513,0.6983172],"study_design_scores_gemma":[0.00008741558,0.00013055936,0.00048153597,0.00028344264,0.00014827818,0.0006673604,0.00008850287,0.2682489,0.07959521,0.58416396,0.06603503,0.00006986367],"about_ca_topic_score_codex":0.000593371,"about_ca_topic_score_gemma":0.0008870714,"teacher_disagreement_score":0.008948051,"about_ca_system_score_codex":0.00055150193,"about_ca_system_score_gemma":0.001092291,"threshold_uncertainty_score":0.029934227},"labels":[],"label_agreement":null},{"id":"W1481028594","doi":"10.1007/3-540-44585-4_1","title":"Software Documentation and the Verification Process","year":2001,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Documentation; Computer science; Software engineering; Notation; Software documentation; Programming language; Software verification; Software; Process (computing); Verification and validation; Formal specification; Software development process; Software development; Software construction; Engineering","score_opus":0.015785737219104094,"score_gpt":0.269389319836322,"score_spread":0.25360358261721794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1481028594","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00973724,0.02958347,0.8217342,0.007761028,0.00076696987,0.00012583367,0.00012510437,0.0015116828,0.12865448],"genre_scores_gemma":[0.43442103,0.02781386,0.4421893,0.0011515862,0.0008295011,0.00034250598,0.00048654785,0.0009894562,0.091776274],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9943,0.002796422,0.00033879082,0.0004529428,0.0018920868,0.00021970905],"domain_scores_gemma":[0.97963434,0.014316657,0.00086062285,0.003134197,0.0018680303,0.000186086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006050372,0.00073798286,0.000765896,0.0023732032,0.0014560928,0.0057567004,0.0013342078,0.002442031,0.007337806],"category_scores_gemma":[0.029075589,0.0010579192,0.0008270729,0.0017764972,0.0058851433,0.008457927,0.002274682,0.0038632057,0.0026990937],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034544228,0.000032366148,0.00021865506,0.0002401788,0.000012186727,0.00012778873,0.00066194363,0.003497143,0.00069453847,0.82438195,0.007366797,0.16273198],"study_design_scores_gemma":[0.000023665374,0.00003064084,0.00017594018,0.00046790592,0.000016536573,0.00024593325,0.00011748354,0.010003812,0.0017639945,0.9157748,0.071356006,0.000023212291],"about_ca_topic_score_codex":0.0021721842,"about_ca_topic_score_gemma":0.0016949397,"teacher_disagreement_score":0.007337806,"about_ca_system_score_codex":0.001598133,"about_ca_system_score_gemma":0.0030131983,"threshold_uncertainty_score":0.03199786},"labels":[],"label_agreement":null},{"id":"W1483143768","doi":"10.1007/3-540-45441-1_15","title":"A UML-Based Approach to System Testing","year":2001,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":137,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Testability; Unified Modeling Language; Sequence diagram; Software engineering; Class diagram; Test case; Model-based testing; Programming language; Reliability engineering; Software; Engineering","score_opus":0.039470597645967595,"score_gpt":0.25275002028524485,"score_spread":0.21327942263927724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1483143768","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006674562,0.00033737088,0.98898125,0.00036603905,0.00006207315,0.000067654204,0.00002672617,0.00085117837,0.008640277],"genre_scores_gemma":[0.03844132,0.0008184659,0.9508404,0.00033995218,0.00011345834,0.0002266311,0.00016365478,0.00045770724,0.008598453],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9947648,0.0020929896,0.00035183816,0.00034428327,0.0022736753,0.00017240737],"domain_scores_gemma":[0.9940951,0.0035541558,0.00020561364,0.0011475035,0.00086837535,0.0001292056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004117427,0.0013486518,0.00088369695,0.0028097748,0.00095126586,0.003478307,0.0033164236,0.0018893902,0.008027661],"category_scores_gemma":[0.012639021,0.0012576076,0.0014583638,0.0017577902,0.002666986,0.0055361073,0.002569625,0.00387436,0.0024939242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004489854,0.0001731334,0.00039374948,0.00035459205,0.000056159595,0.00023616184,0.000710111,0.020494029,0.0055387504,0.6351124,0.010137612,0.3267484],"study_design_scores_gemma":[0.00006136856,0.00008805853,0.00035649832,0.00043123675,0.00010454647,0.0007450612,0.00013918028,0.18503872,0.008613855,0.7003487,0.104018465,0.000054304448],"about_ca_topic_score_codex":0.0021022123,"about_ca_topic_score_gemma":0.0032308977,"teacher_disagreement_score":0.008027661,"about_ca_system_score_codex":0.0012156635,"about_ca_system_score_gemma":0.0012510171,"threshold_uncertainty_score":0.02685523},"labels":[],"label_agreement":null},{"id":"W1484390872","doi":"10.1007/978-3-642-15585-7_12","title":"TeCReVis: A Tool for Test Coverage and Test Redundancy Visualization","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Redundancy (engineering); Computer science; Visualization; Eclipse; Code coverage; Graphical user interface testing; Graphical user interface; Set (abstract data type); Test (biology); Plug-in; Programming language; Software engineering; Data mining; Software; User interface; Operating system; User interface design","score_opus":0.015134570891314921,"score_gpt":0.2711252888983332,"score_spread":0.2559907180070183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1484390872","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005949933,0.00042734956,0.6309614,0.00018428676,0.000102117854,0.00013388965,0.0063156695,0.34718433,0.008740895],"genre_scores_gemma":[0.14108635,0.00086828205,0.75897807,0.00036381153,0.00013249529,0.0007671856,0.019972658,0.06365315,0.014177948],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999141,0.00016303302,0.000084826475,0.00012559528,0.00041544167,0.00007012545],"domain_scores_gemma":[0.99635386,0.0023179709,0.00023509059,0.000526522,0.0004635466,0.00010301119],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013961936,0.0026468278,0.0010817766,0.004664452,0.0005482069,0.002324517,0.0022177938,0.0013304491,0.042435013],"category_scores_gemma":[0.006697437,0.0014546592,0.0013045892,0.0022859168,0.0004004089,0.0025006938,0.0018851488,0.0015364725,0.0064861584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000596667,0.00016886897,0.004126942,0.0016064772,0.00023005217,0.00082211243,0.00082588283,0.03295304,0.019576887,0.02162552,0.24294014,0.6745274],"study_design_scores_gemma":[0.00048723567,0.00021406275,0.0038819634,0.00071375887,0.00029195705,0.0019339473,0.0002476476,0.60351,0.07913247,0.04932836,0.26000318,0.00025546932],"about_ca_topic_score_codex":0.002847336,"about_ca_topic_score_gemma":0.0035175337,"teacher_disagreement_score":0.042435013,"about_ca_system_score_codex":0.0006042329,"about_ca_system_score_gemma":0.0009461692,"threshold_uncertainty_score":0.14195931},"labels":[],"label_agreement":null},{"id":"W1485665058","doi":"10.1007/978-3-540-71289-3_22","title":"A Prioritization Approach for Software Test Cases Based on Bayesian Networks","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":68,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Regression testing; Bayesian network; Computer science; Data mining; Software; Bayesian probability; Reliability engineering; Fault (geology); Fault detection and isolation; Test case; Code coverage; Machine learning; Regression analysis; Artificial intelligence; Software system; Software construction; Engineering; Programming language","score_opus":0.029998499637316784,"score_gpt":0.27108712254050804,"score_spread":0.24108862290319125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1485665058","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036006363,0.00018585472,0.9939421,0.00020645867,0.000023725102,0.00013320443,0.00006136751,0.0005486551,0.0012979193],"genre_scores_gemma":[0.12445968,0.00027901973,0.8724867,0.00015979982,0.00008711562,0.00027963537,0.00028590387,0.00020945238,0.0017526833],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9882321,0.0036018717,0.000806381,0.0012873698,0.005377745,0.0006945431],"domain_scores_gemma":[0.97545755,0.018805701,0.0010627556,0.0010274719,0.0031214203,0.00052512466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009606144,0.0020096526,0.0027312096,0.0069081616,0.0012774628,0.0038135082,0.004282527,0.002330272,0.0057909954],"category_scores_gemma":[0.035721485,0.001999003,0.002350077,0.0033259839,0.0016623583,0.005390833,0.002861402,0.0034414367,0.0009867196],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005374022,0.00045681448,0.0030344392,0.00045706067,0.00033022446,0.00047872192,0.0006075964,0.35891697,0.00679134,0.0982679,0.0061649526,0.52395654],"study_design_scores_gemma":[0.0000581443,0.000043869484,0.0003782183,0.00005970155,0.0001133148,0.00011023698,0.000043597553,0.93512154,0.0016711686,0.06059294,0.0017732476,0.000033976776],"about_ca_topic_score_codex":0.011987115,"about_ca_topic_score_gemma":0.01756675,"teacher_disagreement_score":0.011987115,"about_ca_system_score_codex":0.0030013022,"about_ca_system_score_gemma":0.003720132,"threshold_uncertainty_score":0.050802708},"labels":[],"label_agreement":null},{"id":"W1496915765","doi":"10.1007/11759744_9","title":"Conformance Tests as Checking Experiments for Partial Nondeterministic FSM","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Nondeterministic algorithm; Computer science; Conformance testing; Equivalence (formal languages); Sequence (biology); Automaton; Algorithm; Programming language; Finite-state machine; Model checking; Theoretical computer science; Discrete mathematics; Mathematics; Operating system","score_opus":0.032897483581329594,"score_gpt":0.30610764226534015,"score_spread":0.27321015868401055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1496915765","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.110896744,0.00026228826,0.8756165,0.00029440937,0.0001452059,0.00026658262,0.00028399963,0.0038887528,0.008345594],"genre_scores_gemma":[0.7420076,0.00012435937,0.25263858,0.0001930472,0.00006720185,0.00040190906,0.00039347593,0.00064041984,0.0035334574],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99148566,0.004415453,0.00046315754,0.0009809184,0.0022027602,0.00045194768],"domain_scores_gemma":[0.95674485,0.036617033,0.0010618246,0.0037571788,0.001371568,0.0004475205],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005047162,0.0011257163,0.0010120744,0.0020645375,0.0006306301,0.002131146,0.0022509978,0.0019651598,0.0066721323],"category_scores_gemma":[0.032180645,0.00081774034,0.0013322383,0.0012307762,0.003909164,0.0064018066,0.0023117904,0.0023615675,0.00063551887],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010768252,0.00030863166,0.0031981298,0.0004100707,0.00011349805,0.00045263776,0.000798319,0.08002793,0.036191285,0.7704216,0.0022741223,0.10472704],"study_design_scores_gemma":[0.00013534146,0.0003229124,0.00068145763,0.000076937395,0.00007488442,0.0001863515,0.000110392655,0.4158618,0.045351464,0.53379256,0.0033466665,0.000059249356],"about_ca_topic_score_codex":0.00067844935,"about_ca_topic_score_gemma":0.0009016795,"teacher_disagreement_score":0.0066721323,"about_ca_system_score_codex":0.0011106759,"about_ca_system_score_gemma":0.0007017393,"threshold_uncertainty_score":0.026692212},"labels":[],"label_agreement":null},{"id":"W1498052443","doi":"10.1109/iwsoc.2004.71","title":"Verification strategy determination using dependence analysis of transaction-level models","year":2004,"lang":"en","type":"article","venue":"IEEE International Workshop on System-on-Chip for Real-Time Applications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Bottleneck; Modular design; Transaction-level modeling; Metric (unit); Functional verification; Database transaction; Formal verification; Embedded system; Programming language; Engineering","score_opus":0.09199069560637717,"score_gpt":0.3467204496117136,"score_spread":0.25472975400533643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1498052443","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02765921,0.00007592121,0.97010136,0.000060991784,0.000006784903,0.00012199609,0.0000687462,0.0009474548,0.00095756963],"genre_scores_gemma":[0.6016986,0.00018000918,0.3958185,0.00006171951,0.00002011526,0.00037503368,0.00044270605,0.00026992406,0.001133462],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9929309,0.0025895243,0.0006138426,0.0006199887,0.0029139677,0.00033173626],"domain_scores_gemma":[0.981025,0.010918905,0.0013405464,0.0037799773,0.0027633563,0.00017220018],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004724538,0.0009011208,0.0007274304,0.0032743318,0.0005317618,0.0014513555,0.001576564,0.0009884494,0.0019191169],"category_scores_gemma":[0.023579722,0.00077593833,0.0014343433,0.001131502,0.00086883287,0.0030962552,0.0012961853,0.0012614934,0.00043632276],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047159128,0.00043192197,0.012001999,0.0004033688,0.00021180272,0.00086264615,0.000623463,0.5009545,0.07365968,0.12572959,0.0015772509,0.2830723],"study_design_scores_gemma":[0.000019264218,0.000086149535,0.00041358994,0.000028468115,0.000038991166,0.000106739506,0.000025799793,0.95590585,0.021075256,0.021257212,0.0010187895,0.000023883083],"about_ca_topic_score_codex":0.002053173,"about_ca_topic_score_gemma":0.0024799434,"teacher_disagreement_score":0.004724538,"about_ca_system_score_codex":0.0012180976,"about_ca_system_score_gemma":0.0016663151,"threshold_uncertainty_score":0.024986029},"labels":[],"label_agreement":null},{"id":"W1498361645","doi":"10.1007/11754008_18","title":"Reducing the Lengths of Checking Sequences by Overlapping","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; A priori and a posteriori; Model checking; Set (abstract data type); Selection (genetic algorithm); State (computer science); Algorithm; Reduction (mathematics); Sequence (biology); Theoretical computer science; Artificial intelligence; Mathematics; Programming language","score_opus":0.01894883717841254,"score_gpt":0.25226917097699325,"score_spread":0.23332033379858072,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1498361645","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08911497,0.0005818814,0.8935539,0.00042724708,0.0002349486,0.0003146287,0.0002915206,0.007978796,0.0075020147],"genre_scores_gemma":[0.33907098,0.00041239307,0.6511136,0.00034978273,0.0001472085,0.0003935482,0.00088526146,0.0021103492,0.005516914],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99198043,0.002001268,0.00084191764,0.0015536438,0.0025700126,0.001052681],"domain_scores_gemma":[0.9457991,0.0305469,0.0033564838,0.014694743,0.0044092173,0.0011934949],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033057248,0.0017333868,0.0017230699,0.0022798197,0.0012848758,0.0016107477,0.0043348162,0.001396026,0.010608913],"category_scores_gemma":[0.02466521,0.001760567,0.0018209448,0.0025496557,0.0021324737,0.006158905,0.004059542,0.0033361062,0.0025612204],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029243801,0.0009010117,0.004889167,0.00085701235,0.00017248504,0.0005689084,0.00076394953,0.10625547,0.055549163,0.05787874,0.008761798,0.7604779],"study_design_scores_gemma":[0.0005092205,0.0009486783,0.002358157,0.00034890787,0.00044547717,0.0009190841,0.00044694208,0.6247562,0.111475594,0.23654698,0.021067582,0.00017722476],"about_ca_topic_score_codex":0.0020916655,"about_ca_topic_score_gemma":0.004052151,"teacher_disagreement_score":0.010608913,"about_ca_system_score_codex":0.0012067012,"about_ca_system_score_gemma":0.003322837,"threshold_uncertainty_score":0.035490334},"labels":[],"label_agreement":null},{"id":"W1499663647","doi":"10.11606/t.55.2011.tde-17082011-153853","title":"Contribuições para o Teste de Software","year":2011,"lang":"pt","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Fundação de Amparo à Pesquisa do Estado de São Paulo","keywords":"Software testing; Computer science; Software; Software engineering; Programming language","score_opus":0.046296639843976174,"score_gpt":0.30213648122741865,"score_spread":0.25583984138344246,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1499663647","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055703226,0.13244097,0.40134388,0.030305086,0.011028565,0.00037094665,0.00022179472,0.0017216799,0.36686382],"genre_scores_gemma":[0.58731335,0.14210647,0.18364972,0.003090284,0.0146753285,0.00055711303,0.00042181264,0.0012381282,0.066947915],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98626596,0.005515255,0.00050100323,0.0013966722,0.005904682,0.00041645663],"domain_scores_gemma":[0.92242664,0.06288488,0.0016743519,0.0048435023,0.0072051547,0.000965386],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008122753,0.0016694957,0.0012192003,0.005936132,0.0024929103,0.0076723625,0.0019831823,0.0025752555,0.0074309576],"category_scores_gemma":[0.058383223,0.0008861594,0.0012694739,0.0043676025,0.0062969727,0.0063584256,0.0028608334,0.0063745365,0.0028290611],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015607302,0.00023365696,0.0029074198,0.0022112294,0.00009288658,0.0009879877,0.0042228503,0.0059765484,0.0049577956,0.32132515,0.012827429,0.644101],"study_design_scores_gemma":[0.00006832426,0.00047846112,0.0053023216,0.0023997643,0.00020852864,0.006185014,0.0021692947,0.034417927,0.016880335,0.46495655,0.46677926,0.00015426581],"about_ca_topic_score_codex":0.0016009429,"about_ca_topic_score_gemma":0.0017594377,"teacher_disagreement_score":0.008122753,"about_ca_system_score_codex":0.0024877328,"about_ca_system_score_gemma":0.0018060439,"threshold_uncertainty_score":0.042957783},"labels":[],"label_agreement":null},{"id":"W1501211454","doi":"10.1007/978-90-481-3658-2_79","title":"Testing Grammars For Top-Down Parsers","year":2009,"lang":"en","type":"book-chapter","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Programming language; Compiler; Parsing; Rule-based machine translation; Compiler construction; Grammar; Implementation; Natural language processing; Linguistics","score_opus":0.053056053212618005,"score_gpt":0.2653031525672882,"score_spread":0.2122470993546702,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1501211454","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010753132,0.0012439555,0.94930243,0.00060022215,0.00017578465,0.00012526117,0.0002962582,0.007523039,0.02997994],"genre_scores_gemma":[0.2517514,0.0018826358,0.7071836,0.0005008676,0.00018685506,0.00024154094,0.0021345166,0.0048221857,0.031296395],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99606884,0.0010376056,0.0003079913,0.0006589909,0.0016983795,0.00022814638],"domain_scores_gemma":[0.98800397,0.008632254,0.00016096151,0.0020647643,0.0010355583,0.00010254858],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022984464,0.0015104961,0.0014685666,0.001677927,0.0009580024,0.0030965982,0.003236877,0.0017439713,0.011282233],"category_scores_gemma":[0.014919617,0.0013476773,0.0019478005,0.0016334793,0.0034354778,0.0058181887,0.0025855296,0.003869891,0.0037462604],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009415332,0.00012152837,0.0008757465,0.00058871816,0.000055326447,0.0006034303,0.0012941521,0.015595827,0.01366758,0.33422667,0.017783934,0.61509293],"study_design_scores_gemma":[0.00004552436,0.000071981485,0.00048211307,0.00039884026,0.000107290565,0.0010085551,0.00021416786,0.08873343,0.035749067,0.8172278,0.05587843,0.000082745086],"about_ca_topic_score_codex":0.0016998014,"about_ca_topic_score_gemma":0.0015746129,"teacher_disagreement_score":0.011282233,"about_ca_system_score_codex":0.0013331325,"about_ca_system_score_gemma":0.0010293297,"threshold_uncertainty_score":0.037742853},"labels":[],"label_agreement":null},{"id":"W1503243085","doi":"","title":"SCL: a language for security testing of network applications","year":2005,"lang":"en","type":"article","venue":"Conference of the Centre for Advanced Studies on Collaborative Research","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Royal Military College of Canada","funders":"","keywords":"Computer science; Syntax; Programming language; Abstract syntax tree; Abstract syntax; Protocol (science); Artificial intelligence","score_opus":0.1341926941477668,"score_gpt":0.4334524182713786,"score_spread":0.2992597241236118,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1503243085","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018738585,0.00019841216,0.935042,0.000597815,0.00011671949,0.00049961626,0.0036761274,0.050916847,0.0070786527],"genre_scores_gemma":[0.058772612,0.0008233215,0.88946134,0.0016015181,0.00026187958,0.002902844,0.014548382,0.01723815,0.014389945],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9948027,0.0015700364,0.0011119738,0.0005093308,0.0016579923,0.00034797686],"domain_scores_gemma":[0.99113303,0.005061804,0.00078197056,0.0010796217,0.0016442401,0.00029945408],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048432625,0.0024780487,0.0012081964,0.003228431,0.0012028923,0.004251471,0.0039434526,0.0024952474,0.016971573],"category_scores_gemma":[0.010044372,0.0020254354,0.00269062,0.0022803412,0.0033674126,0.005269803,0.0025170338,0.0047842474,0.008051218],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005018877,0.00034011633,0.0021070386,0.0029898556,0.00015612016,0.002113806,0.0029833731,0.044946898,0.023511728,0.4826909,0.22650434,0.21115392],"study_design_scores_gemma":[0.00033997232,0.00027729612,0.00064070267,0.0008623838,0.0000877458,0.002477088,0.00032154474,0.14553131,0.020164879,0.20551617,0.62353927,0.00024155827],"about_ca_topic_score_codex":0.005255555,"about_ca_topic_score_gemma":0.0045425417,"teacher_disagreement_score":0.016971573,"about_ca_system_score_codex":0.001875082,"about_ca_system_score_gemma":0.0041705733,"threshold_uncertainty_score":0.05677551},"labels":[],"label_agreement":null},{"id":"W1504886604","doi":"10.1007/0-306-47003-9_6","title":"Diagnosing Multiple Faults in Communicating Finite State Machines","year":2006,"lang":"en","type":"book-chapter","venue":"Kluwer Academic Publishers eBooks","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Finite-state machine; Computer science; Component (thermodynamics); State (computer science); Abstract state machines; Extended finite-state machine; Fault (geology); Distributed computing; Algorithm","score_opus":0.029973827013449045,"score_gpt":0.26035638916998355,"score_spread":0.2303825621565345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1504886604","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014198892,0.0005555161,0.98106164,0.0001803703,0.00007017136,0.000040663457,0.000019119218,0.00095786963,0.0029158026],"genre_scores_gemma":[0.2970002,0.0008571846,0.69556963,0.00015847213,0.00008469662,0.000100205,0.000118319454,0.00022901568,0.0058823363],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99846506,0.0004382905,0.00007381179,0.00023613329,0.00070744223,0.000079296326],"domain_scores_gemma":[0.9944102,0.004769414,0.0001674886,0.0003756108,0.00023076712,0.00004646655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001184144,0.00081103505,0.00071010203,0.0009740392,0.0005284356,0.0010904247,0.0016799795,0.0013413018,0.0024534632],"category_scores_gemma":[0.005240151,0.00051227474,0.00064809335,0.0005635367,0.0018497206,0.0024492564,0.0011797632,0.0020334783,0.0005263301],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022081529,0.00012868734,0.0016439477,0.0008037902,0.00006951907,0.002169267,0.0016089086,0.1043849,0.041704442,0.31078967,0.0031741518,0.53330195],"study_design_scores_gemma":[0.0000569478,0.0002334794,0.0005397863,0.0003144448,0.000075062075,0.0023403962,0.0001888904,0.54064727,0.0829055,0.3426436,0.029977169,0.0000775345],"about_ca_topic_score_codex":0.00041413747,"about_ca_topic_score_gemma":0.00046985823,"teacher_disagreement_score":0.0024534632,"about_ca_system_score_codex":0.0006442675,"about_ca_system_score_gemma":0.0006534935,"threshold_uncertainty_score":0.008207679},"labels":[],"label_agreement":null},{"id":"W1505714538","doi":"10.1007/978-3-540-31810-1_10","title":"Applying Reduction Techniques to Software Functional Requirement Specifications","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Program slicing; Computer science; Slicing; Reuse; Functional requirement; Programming language; Non-functional requirement; Software requirements specification; Software engineering; Program comprehension; Software maintenance; Notation; Functional specification; Formal specification; Software development; Reliability engineering; Software; Software system; Software construction","score_opus":0.060229339611306124,"score_gpt":0.2776802005003229,"score_spread":0.21745086088901677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1505714538","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0057116887,0.00017501017,0.9835175,0.00016200666,0.00004155342,0.000086917025,0.00007416562,0.0013088327,0.008922212],"genre_scores_gemma":[0.15562843,0.00071995606,0.8290214,0.00023296931,0.00008353146,0.00027241156,0.00094081933,0.0012940579,0.011806451],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9984017,0.00047123633,0.0001132827,0.00013119687,0.0007836726,0.00009893309],"domain_scores_gemma":[0.9971251,0.0019566347,0.00007224323,0.00046158192,0.0003649384,0.000019505105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010967709,0.0008166575,0.000631979,0.0016328138,0.00056075567,0.000897788,0.0013516736,0.00060958293,0.0056188456],"category_scores_gemma":[0.004839743,0.00083949877,0.0020388423,0.0011216865,0.0012968206,0.0015877994,0.0012712816,0.0022533648,0.0015691529],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011770131,0.00015267443,0.0004361687,0.0008108621,0.00009276525,0.00043441373,0.0007221758,0.054986443,0.024353946,0.2797035,0.01227585,0.6259136],"study_design_scores_gemma":[0.00008083422,0.00013737965,0.0006664164,0.0001964224,0.000139818,0.00063017884,0.00023611712,0.3487842,0.036326144,0.558327,0.054398134,0.00007740262],"about_ca_topic_score_codex":0.002947237,"about_ca_topic_score_gemma":0.0038158929,"teacher_disagreement_score":0.0056188456,"about_ca_system_score_codex":0.00064323423,"about_ca_system_score_gemma":0.0007240563,"threshold_uncertainty_score":0.01879692},"labels":[],"label_agreement":null},{"id":"W1514771780","doi":"10.1007/978-3-642-20398-5_38","title":"A Tabular Expression Toolbox for Matlab/Simulink","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Toolbox; Computer science; MATLAB; Solver; Programming language; Interface (matter); Expression (computer science); Code (set theory); Completeness (order theory); Software engineering; Embedded system; Operating system","score_opus":0.03202961409431724,"score_gpt":0.2631017049716519,"score_spread":0.23107209087733468,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1514771780","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006673761,0.00044497952,0.882536,0.000108005224,0.00021933942,0.00014685585,0.004663437,0.09346109,0.01775281],"genre_scores_gemma":[0.028678771,0.0018655154,0.7914379,0.00065675005,0.00018061229,0.0012589312,0.017105808,0.066095866,0.09271983],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993895,0.000139023,0.000082696926,0.00007997326,0.00025362472,0.00005509959],"domain_scores_gemma":[0.9984408,0.0006674492,0.00008522898,0.00023290321,0.00053091726,0.00004264016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081990275,0.0018792712,0.00095726835,0.0013549821,0.00040244742,0.0014407049,0.0023223658,0.0009159649,0.16358876],"category_scores_gemma":[0.0033443624,0.000960399,0.0009534673,0.0012254245,0.00034086825,0.0013620023,0.0009724151,0.0019594978,0.10239551],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044152036,0.00017907984,0.00039594882,0.002217555,0.00009244325,0.00047775352,0.00025896527,0.023413988,0.02384754,0.037629075,0.3581534,0.5528928],"study_design_scores_gemma":[0.00030598362,0.00017201339,0.0005847359,0.0005890975,0.0001099919,0.0013112035,0.0000892655,0.108608246,0.046518996,0.035660584,0.80591625,0.00013366579],"about_ca_topic_score_codex":0.0009613603,"about_ca_topic_score_gemma":0.001264366,"teacher_disagreement_score":0.16358876,"about_ca_system_score_codex":0.0003531084,"about_ca_system_score_gemma":0.0006941272,"threshold_uncertainty_score":0.547259},"labels":[],"label_agreement":null},{"id":"W1520324981","doi":"10.1007/11430230_8","title":"Multiplexing of Partially Ordered Events","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal; McGill University","funders":"","keywords":"Multiplexer; Merge (version control); Computer science; Multiplexing; TRACE (psycholinguistics); Distributed computing; Code (set theory); Algorithm; Theoretical computer science; Parallel computing; Programming language; Telecommunications","score_opus":0.02697436365411322,"score_gpt":0.2704738527400422,"score_spread":0.243499489085929,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1520324981","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06517746,0.0003746071,0.91286874,0.00012180398,0.00028223536,0.0000909713,0.00022077761,0.003107393,0.017756036],"genre_scores_gemma":[0.7781036,0.0004436951,0.19972111,0.00013861195,0.00025471358,0.00012857368,0.0004998127,0.0005325725,0.020177407],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985958,0.0002280543,0.00009631907,0.00030246767,0.0005183658,0.00025912272],"domain_scores_gemma":[0.9968305,0.001315547,0.00025842155,0.0008632423,0.00038661997,0.00034560214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011023076,0.0010074242,0.00081675244,0.0014839069,0.0008396359,0.0021410608,0.0010640464,0.00051274424,0.009774668],"category_scores_gemma":[0.0046874834,0.0006495905,0.00066723087,0.0011264038,0.00069779635,0.0037882242,0.0020088807,0.0011117735,0.0014945925],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001535057,0.00026905883,0.0022759687,0.0002561454,0.00010017597,0.0010448544,0.00042425893,0.029470323,0.07065551,0.61893106,0.004770224,0.2702673],"study_design_scores_gemma":[0.000088734545,0.00046407196,0.0012382808,0.00008092181,0.00015877285,0.0013379167,0.00019257411,0.30901024,0.113613665,0.54564285,0.02804214,0.00012986276],"about_ca_topic_score_codex":0.00037225743,"about_ca_topic_score_gemma":0.00046302576,"teacher_disagreement_score":0.009774668,"about_ca_system_score_codex":0.000512785,"about_ca_system_score_gemma":0.0005787649,"threshold_uncertainty_score":0.032699466},"labels":[],"label_agreement":null},{"id":"W1520613697","doi":"10.1007/11754008_14","title":"Detecting Observability Problems in Distributed Testing","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; University of Windsor","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Observability; Controllability; Sequence (biology); Computer science; Construct (python library); State (computer science); Extension (predicate logic); Sequence diagram; Finite-state machine; Theoretical computer science; Algorithm; Test (biology); Distributed computing; Unified Modeling Language; Programming language; Mathematics; Software","score_opus":0.04026543076394748,"score_gpt":0.2534101892339219,"score_spread":0.2131447584699744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1520613697","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14404249,0.0005662859,0.8499124,0.0003254466,0.000043348788,0.00010005268,0.00007454255,0.0025329788,0.0024024716],"genre_scores_gemma":[0.8390993,0.00016743063,0.15867716,0.00007323899,0.000041408173,0.000064256616,0.00018543613,0.00022165624,0.0014702294],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99464595,0.0015749293,0.00032332973,0.00087484304,0.0020568077,0.0005240838],"domain_scores_gemma":[0.92539084,0.06216171,0.0034744805,0.006115508,0.0021927655,0.0006646781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033535538,0.0010183423,0.0012437168,0.0015856366,0.0004937944,0.0013266469,0.0024809758,0.0015967817,0.0013399962],"category_scores_gemma":[0.03324938,0.0009108685,0.0010513534,0.001100721,0.0021488285,0.004620718,0.0024546972,0.0026052946,0.00018918939],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001623997,0.0008253258,0.0513768,0.00066596095,0.00022202497,0.0016790216,0.0010760131,0.27306294,0.027193448,0.055988397,0.0033548905,0.5829312],"study_design_scores_gemma":[0.000098100994,0.00030722693,0.0049588853,0.00007285124,0.000095555435,0.0008906377,0.00017694641,0.82054687,0.01984049,0.15169838,0.0012730111,0.000041082127],"about_ca_topic_score_codex":0.0014161537,"about_ca_topic_score_gemma":0.0019312596,"teacher_disagreement_score":0.0033535538,"about_ca_system_score_codex":0.0007223441,"about_ca_system_score_gemma":0.00073650293,"threshold_uncertainty_score":0.017735481},"labels":[],"label_agreement":null},{"id":"W1522741019","doi":"10.1002/stvr.1572","title":"Coverage‐based regression test case selection, minimization and prioritization: a case study on an industrial system","year":2015,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":95,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg","keywords":"Regression testing; Minification; Selection (genetic algorithm); Computer science; Prioritization; Fault detection and isolation; Suite; Reliability engineering; Test suite; Regression; Fault (geology); Regression analysis; Data mining; Machine learning; Test case; Artificial intelligence; Statistics; Engineering; Software; Mathematics; Software system","score_opus":0.08433280682020698,"score_gpt":0.31182946332980477,"score_spread":0.22749665650959777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1522741019","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9730856,0.00013677601,0.02413686,0.00019162367,0.00000695634,0.00016013047,0.00007457532,0.00026397375,0.0019434575],"genre_scores_gemma":[0.97401583,0.000071771894,0.02511113,0.000025665748,0.0000045079405,0.00005683756,0.000075129974,0.000029043966,0.0006099965],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977835,0.0010923035,0.0001004659,0.0001448867,0.0007050007,0.00017380404],"domain_scores_gemma":[0.985876,0.011496888,0.00071484956,0.0006826772,0.0010271533,0.00020235786],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023628203,0.00047692825,0.00037538438,0.0011631858,0.0005224155,0.00047363777,0.0009475021,0.0007806709,0.0010363775],"category_scores_gemma":[0.009054066,0.00026103697,0.00043809277,0.0008853189,0.0005641717,0.0004912508,0.00038070767,0.0004981832,0.00014136282],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016223679,0.0025813682,0.05502882,0.0008289422,0.00023156728,0.010983381,0.0018961363,0.58164716,0.06314454,0.0049145604,0.0029322458,0.2741889],"study_design_scores_gemma":[0.000512466,0.0038626858,0.029809136,0.000075340235,0.00022277806,0.0036294411,0.0010607197,0.8720822,0.08189059,0.0016455263,0.00513286,0.00007624548],"about_ca_topic_score_codex":0.007404279,"about_ca_topic_score_gemma":0.008795127,"teacher_disagreement_score":0.007404279,"about_ca_system_score_codex":0.0009127594,"about_ca_system_score_gemma":0.00076461985,"threshold_uncertainty_score":0.014722347},"labels":[],"label_agreement":null},{"id":"W1527815941","doi":"10.1002/stvr.1576","title":"Automatic fault localization for client‐side JavaScript","year":2015,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Intel Corporation","keywords":"JavaScript; Unobtrusive JavaScript; Computer science; Ajax; Scripting language; Web application; Debugging; Programming language; Rich Internet application; Document Object Model; Code (set theory); Source code; Dynamic web page; Tracing; Operating system; World Wide Web; Web service; Web page","score_opus":0.05771615534496858,"score_gpt":0.29183010284763583,"score_spread":0.23411394750266726,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1527815941","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19043401,0.000654395,0.7327089,0.00020089731,0.000070735645,0.00012118957,0.00027044286,0.07363682,0.0019026608],"genre_scores_gemma":[0.8022486,0.00012168762,0.19386353,0.00007133304,0.000019596748,0.0000553434,0.0006019349,0.0013279072,0.0016900374],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99803585,0.00041148937,0.00013501669,0.0004109276,0.00087246316,0.00013430754],"domain_scores_gemma":[0.9923896,0.002901183,0.0011705635,0.0015311972,0.0018293151,0.00017812384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009757125,0.0009826102,0.00072305615,0.0018284121,0.0004180492,0.0007723214,0.0013455887,0.0007999582,0.001384773],"category_scores_gemma":[0.005944906,0.0004004349,0.00057361746,0.0005100758,0.0005181585,0.00085974886,0.00081776566,0.00073483086,0.0009633389],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046624022,0.0002354799,0.022451712,0.0005634383,0.00009094829,0.00093860226,0.00066442485,0.07010558,0.19205551,0.0032375741,0.007683363,0.70150703],"study_design_scores_gemma":[0.0000352618,0.00016769748,0.0058811023,0.00006868567,0.000043155644,0.00085647526,0.00008698007,0.82660633,0.1585603,0.00305994,0.004590109,0.000043889795],"about_ca_topic_score_codex":0.0027876534,"about_ca_topic_score_gemma":0.0019579085,"teacher_disagreement_score":0.0027876534,"about_ca_system_score_codex":0.000668508,"about_ca_system_score_gemma":0.0010163786,"threshold_uncertainty_score":0.0055428743},"labels":[],"label_agreement":null},{"id":"W1536265389","doi":"10.1007/3-540-46423-9_2","title":"Optimizing Java Bytecode Using the Soot Framework: Is It Feasible?","year":2000,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":313,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Bytecode; Java bytecode; Computer science; Java; Programming language; Java annotation; Class (philosophy); Generics in Java; Program optimization; Java applet; Operating system; Compiler; Artificial intelligence","score_opus":0.05289934935660727,"score_gpt":0.30319578851342266,"score_spread":0.2502964391568154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1536265389","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.102399394,0.003654298,0.83425856,0.0035608197,0.0005343759,0.00008674145,0.00020554457,0.026775414,0.02852483],"genre_scores_gemma":[0.5190296,0.0020888224,0.45905653,0.00064208533,0.00015870295,0.00008102616,0.00050336757,0.006222447,0.012217349],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99861526,0.00031692337,0.000056121175,0.00018160178,0.0005206033,0.00030956444],"domain_scores_gemma":[0.99727684,0.0009661342,0.00017332293,0.0009676834,0.0005208286,0.000095184216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011519503,0.0009612363,0.0010742081,0.0004554459,0.00043383948,0.0015864681,0.0017244107,0.00096383796,0.005710925],"category_scores_gemma":[0.0052539846,0.0005940288,0.00091968034,0.00070184993,0.00087808596,0.0048975237,0.0007802619,0.0015451772,0.0018764373],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009462987,0.00029500376,0.004262058,0.0005423957,0.00017498924,0.0001403301,0.00020164085,0.05462862,0.030804036,0.10709894,0.024142789,0.7767628],"study_design_scores_gemma":[0.00025907747,0.00047279685,0.0032756007,0.000291982,0.00031474076,0.00039419375,0.00032516767,0.636728,0.078263976,0.21282609,0.06670976,0.00013861124],"about_ca_topic_score_codex":0.003691611,"about_ca_topic_score_gemma":0.005886809,"teacher_disagreement_score":0.005710925,"about_ca_system_score_codex":0.00052595796,"about_ca_system_score_gemma":0.001530004,"threshold_uncertainty_score":0.019104958},"labels":[],"label_agreement":null},{"id":"W1537009764","doi":"10.1023/a:1021634422504","title":"Specification-based Testing for Gui-based Applications","year":2002,"lang":"en","type":"article","venue":"Software Quality Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Programming language; Java; Formal specification; Test harness; Finite-state machine; Graphical user interface testing; Test case; Implementation; Software engineering; Keyword-driven testing; Graphical user interface; Software; User interface; Software development","score_opus":0.15669253087361484,"score_gpt":0.34311759476107645,"score_spread":0.1864250638874616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1537009764","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08443998,0.0002099128,0.90920925,0.00022665148,0.000027081382,0.000105036066,0.00006280753,0.003925744,0.0017935943],"genre_scores_gemma":[0.7707315,0.00016002901,0.22663471,0.00012998212,0.000016458605,0.000121906654,0.00039025812,0.0005181055,0.0012970802],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9917468,0.003893953,0.0005416868,0.00047186489,0.0028192815,0.0005263248],"domain_scores_gemma":[0.96368086,0.025857164,0.0014210842,0.004634395,0.0039638756,0.00044256973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054929513,0.0008168603,0.0006877073,0.0012634395,0.0005354482,0.0015794631,0.002686498,0.001520533,0.002552784],"category_scores_gemma":[0.027141431,0.00060969766,0.0009058754,0.0008836767,0.0011878381,0.002482311,0.0012744301,0.0016151698,0.00054338184],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023094683,0.001657608,0.021104006,0.00091909256,0.00022895145,0.0015771616,0.0015261712,0.20680846,0.11059142,0.0969352,0.0068290676,0.54951346],"study_design_scores_gemma":[0.00017314433,0.000539932,0.0017806689,0.00007252803,0.00009616488,0.0004973882,0.00013023261,0.9248952,0.048129056,0.021856459,0.0017926267,0.000036635938],"about_ca_topic_score_codex":0.0038927067,"about_ca_topic_score_gemma":0.0040530595,"teacher_disagreement_score":0.0054929513,"about_ca_system_score_codex":0.00074179604,"about_ca_system_score_gemma":0.0013866661,"threshold_uncertainty_score":0.029049873},"labels":[],"label_agreement":null},{"id":"W1537363352","doi":"10.11606/d.45.2008.tde-11082008-134008","title":"Ambiente de testes utilizando verificação de componentes java com tratamento de exceções","year":2008,"lang":"pt","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institute of Gender and Health","funders":"","keywords":"Computer science; Maintainability; Java; Programming language; Source lines of code; Robustness (evolution); Software; Operating system; Reliability engineering; Software engineering","score_opus":0.04053729441339841,"score_gpt":0.30015764380422694,"score_spread":0.2596203493908285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1537363352","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5609147,0.0036595224,0.37298566,0.0010623169,0.0007583654,0.0009042349,0.0008563483,0.029129202,0.02972966],"genre_scores_gemma":[0.79294854,0.001931932,0.1936917,0.00021288382,0.00026643288,0.00031272715,0.001180909,0.0020869295,0.007367931],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99001807,0.0025932163,0.0004217284,0.0017937246,0.004698239,0.00047508394],"domain_scores_gemma":[0.96463436,0.020443559,0.0015270365,0.005930489,0.0067857034,0.0006788573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007848233,0.0013312731,0.0013394865,0.0023497674,0.0014047045,0.004587638,0.0017622052,0.0015479303,0.0047674943],"category_scores_gemma":[0.039699174,0.00074620143,0.0012539894,0.0017006504,0.0016117878,0.0029429959,0.0023266776,0.0019774428,0.0014669844],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002706846,0.0014843583,0.01655942,0.0028122326,0.00032087418,0.0033104988,0.0076561873,0.023972293,0.29166177,0.019400815,0.008467662,0.621647],"study_design_scores_gemma":[0.0006682318,0.0024516652,0.03285582,0.0008536079,0.00068809453,0.0022568617,0.0030116523,0.31643328,0.5320468,0.017876606,0.090516046,0.00034140307],"about_ca_topic_score_codex":0.008037216,"about_ca_topic_score_gemma":0.007090259,"teacher_disagreement_score":0.008037216,"about_ca_system_score_codex":0.0010859236,"about_ca_system_score_gemma":0.0013026462,"threshold_uncertainty_score":0.041505933},"labels":[],"label_agreement":null},{"id":"W1540848743","doi":"10.1007/978-3-540-70567-3_21","title":"Towards Automation of Testing High-Level Security Properties","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Security testing; Software security assurance; Computer science; System integration testing; Computer security; Software reliability testing; Automation; Software; Argument (complex analysis); Software testing; Software engineering; Software development; Software construction; Engineering; Cloud computing security; Security information and event management; Information security; Security service; Operating system; Cloud computing","score_opus":0.05668893060895422,"score_gpt":0.2531988604468793,"score_spread":0.19650992983792506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1540848743","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074863727,0.00024934235,0.9827626,0.00016999243,0.000043582957,0.0000684315,0.00006975837,0.0053809355,0.0037690043],"genre_scores_gemma":[0.13016716,0.00055749883,0.8638897,0.00024086302,0.00007644752,0.000119322656,0.00049641065,0.00083199056,0.0036206322],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9930012,0.0021422033,0.00048220952,0.0008860381,0.003066805,0.0004213948],"domain_scores_gemma":[0.98082113,0.010530419,0.0006175542,0.006084425,0.0017346663,0.00021178945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027765778,0.0014329467,0.0011691967,0.0015320429,0.0005234233,0.0027133527,0.002323587,0.0014739829,0.0047625126],"category_scores_gemma":[0.010513226,0.0015358398,0.0018710884,0.0009920034,0.0018557506,0.0035829777,0.0026981323,0.0057186284,0.0033440972],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034259618,0.00058921415,0.003215324,0.0011860059,0.000162412,0.00066292187,0.0009986861,0.03348495,0.09424993,0.13255884,0.011440354,0.7211088],"study_design_scores_gemma":[0.00016811752,0.00035093745,0.0025186217,0.00041060688,0.00023079748,0.0019352286,0.00015719626,0.36993286,0.19883901,0.37503728,0.050327398,0.00009190687],"about_ca_topic_score_codex":0.00086299994,"about_ca_topic_score_gemma":0.0012360634,"teacher_disagreement_score":0.0047625126,"about_ca_system_score_codex":0.0005956511,"about_ca_system_score_gemma":0.0013081633,"threshold_uncertainty_score":0.015932143},"labels":[],"label_agreement":null},{"id":"W1546593066","doi":"10.1109/ccece.2001.933644","title":"Component interaction testing using model-checking","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Component (thermodynamics); Model checking; Model-based testing; Programming language; Simple (philosophy); Semantics (computer science); Test case; Formal semantics (linguistics); Object-oriented programming; Formal specification; Formal methods; Theoretical computer science; Software engineering; Machine learning","score_opus":0.2274275624985504,"score_gpt":0.31140505006873614,"score_spread":0.08397748757018575,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1546593066","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008197887,0.00009436879,0.9870526,0.00012430976,0.000022910282,0.000071595285,0.00003996573,0.0027944078,0.0016017837],"genre_scores_gemma":[0.3094322,0.00029229786,0.6873094,0.00014235784,0.000032998567,0.0003199435,0.0004261567,0.00066730974,0.0013773631],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98788303,0.005795281,0.0005610509,0.00094432203,0.004315321,0.0005009468],"domain_scores_gemma":[0.98064226,0.012971985,0.00072397577,0.0037932424,0.0016911849,0.00017732815],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006784426,0.0014340536,0.0010669896,0.0022671053,0.0007034943,0.00262239,0.003093516,0.0018143264,0.0028164203],"category_scores_gemma":[0.025535522,0.0008860653,0.0022456655,0.0014932049,0.0022468676,0.004526674,0.002256299,0.0021756885,0.00057397265],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037952707,0.0005406355,0.005362004,0.0005622829,0.00040039912,0.0005671103,0.00045862753,0.5337317,0.028683878,0.17839058,0.0051952978,0.2457279],"study_design_scores_gemma":[0.00009486001,0.00010953999,0.00026552338,0.0000632788,0.0000643522,0.00017192368,0.000020238856,0.9247739,0.019552074,0.050960965,0.0038920948,0.000031313804],"about_ca_topic_score_codex":0.0050380374,"about_ca_topic_score_gemma":0.0036643022,"teacher_disagreement_score":0.006784426,"about_ca_system_score_codex":0.0016406728,"about_ca_system_score_gemma":0.002429216,"threshold_uncertainty_score":0.03587991},"labels":[],"label_agreement":null},{"id":"W1550040990","doi":"10.1007/978-3-642-05031-2_4","title":"Implementing MSC Tests with Quiescence Observation","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Asynchronous communication; Test (biology); Chart; Fault detection and isolation; Sequence (biology); Distributed computing; Fault (geology); Test case; Class (philosophy); Reliability engineering; Real-time computing; Embedded system; Programming language; Artificial intelligence; Computer network; Machine learning; Engineering","score_opus":0.02877733741771017,"score_gpt":0.26947884009723283,"score_spread":0.24070150267952267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1550040990","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07436527,0.00010189796,0.8793693,0.0002766225,0.00013702014,0.00014198301,0.00024522597,0.032954246,0.012408504],"genre_scores_gemma":[0.75938594,0.00006413297,0.23044457,0.00018911416,0.000072542825,0.00015105157,0.0005344804,0.002802513,0.0063556163],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99562645,0.0010821597,0.00031183174,0.00050785125,0.0018278391,0.0006439395],"domain_scores_gemma":[0.98301226,0.008275719,0.0010688497,0.004746836,0.0024551875,0.00044103822],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002294127,0.0011804511,0.0007376654,0.0012752509,0.0005105403,0.0022119116,0.0027575411,0.0013292257,0.008908634],"category_scores_gemma":[0.016718252,0.0007567988,0.0007506619,0.00090449455,0.0015620594,0.0040816274,0.002640727,0.0015411202,0.0023318003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028227665,0.0009746051,0.01543418,0.00057449384,0.00017296505,0.0011705622,0.0015440314,0.061242554,0.083236426,0.13926934,0.018675666,0.6748824],"study_design_scores_gemma":[0.00036030193,0.0006583973,0.0019913889,0.00013313082,0.00020942459,0.00051229866,0.00027940844,0.6182138,0.24726824,0.1021719,0.028090239,0.000111563226],"about_ca_topic_score_codex":0.0022356533,"about_ca_topic_score_gemma":0.0025406543,"teacher_disagreement_score":0.008908634,"about_ca_system_score_codex":0.0007855219,"about_ca_system_score_gemma":0.0013625188,"threshold_uncertainty_score":0.029802322},"labels":[],"label_agreement":null},{"id":"W1553735400","doi":"10.1007/11888116_32","title":"Minimizing Coordination Channels in Distributed Testing","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Sabancı Üniversitesi; Government of Ontario","keywords":"Computer science; Architecture; Distributed computing; Test (biology); Test case; Transitive relation; Systems architecture","score_opus":0.028546060329400658,"score_gpt":0.25505638375437323,"score_spread":0.22651032342497257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1553735400","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029536167,0.0003199688,0.9628964,0.00033409006,0.00005303457,0.00012175122,0.000038541148,0.0011627496,0.005537366],"genre_scores_gemma":[0.7047268,0.00029848472,0.287996,0.0001460509,0.00009701235,0.0002860304,0.00011997,0.0005954266,0.0057342197],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.992804,0.003260876,0.00037823195,0.0009427211,0.0016448955,0.00096931454],"domain_scores_gemma":[0.95679986,0.031131892,0.0022373064,0.0068645184,0.0016725191,0.001293863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005668322,0.0014794109,0.0022493878,0.0010829896,0.001376355,0.0019433822,0.0034617123,0.0015627071,0.0054380037],"category_scores_gemma":[0.023000548,0.0010186353,0.0010708418,0.0014747116,0.0028702181,0.0055193733,0.0041313795,0.0034950022,0.00063792436],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001492078,0.00051355304,0.002653013,0.00049788394,0.00011015704,0.0003401042,0.00048516845,0.42791483,0.008048925,0.24700043,0.0058124964,0.30513135],"study_design_scores_gemma":[0.00022139715,0.00026412393,0.00048595323,0.00007891651,0.000118649084,0.00020591445,0.000116622825,0.64262915,0.0068927566,0.34599283,0.0029566644,0.000037014706],"about_ca_topic_score_codex":0.0015537973,"about_ca_topic_score_gemma":0.0020655664,"teacher_disagreement_score":0.005668322,"about_ca_system_score_codex":0.0015335958,"about_ca_system_score_gemma":0.0024845744,"threshold_uncertainty_score":0.029977322},"labels":[],"label_agreement":null},{"id":"W1558340936","doi":"10.1109/iccad.1988.122492","title":"Parallel PLA fault simulation based on Boolean vector operations","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Set (abstract data type); Computer science; Representation (politics); Fault (geology); Bitwise operation; Parallel computing; Theoretical computer science; Algorithm; Programming language","score_opus":0.03182257309000146,"score_gpt":0.2904699293622739,"score_spread":0.2586473562722724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1558340936","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03705239,0.00006154559,0.95162374,0.00007455871,0.00003054798,0.00006579143,0.00012185225,0.005512822,0.005456713],"genre_scores_gemma":[0.61089766,0.00013973599,0.3832599,0.000051186104,0.000015120831,0.00016612663,0.00038351337,0.00037503772,0.0047116685],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99971646,0.000054717068,0.000017492179,0.00005143959,0.00012709791,0.000032767133],"domain_scores_gemma":[0.9994461,0.00025274736,0.00006136492,0.000098510856,0.00012167588,0.000019663972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029305584,0.0006795485,0.00044142478,0.00054780813,0.0003734982,0.0006768016,0.000970989,0.00042612478,0.0060186135],"category_scores_gemma":[0.0011061782,0.00026288844,0.00036795996,0.00050956797,0.00047860286,0.0013095832,0.00054123055,0.00048573306,0.0007782376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037503403,0.000104706145,0.0012533007,0.00013909118,0.000030498091,0.00016401173,0.0000981938,0.7956014,0.037715394,0.030318296,0.0028149718,0.13138513],"study_design_scores_gemma":[0.000012582403,0.00003957883,0.000053890126,0.00000288555,0.000004838889,0.000025084117,0.0000055932296,0.9847035,0.010347683,0.0038087917,0.0009904451,0.0000050687768],"about_ca_topic_score_codex":0.005292538,"about_ca_topic_score_gemma":0.005234003,"teacher_disagreement_score":0.0060186135,"about_ca_system_score_codex":0.00080467877,"about_ca_system_score_gemma":0.00093106227,"threshold_uncertainty_score":0.02013427},"labels":[],"label_agreement":null},{"id":"W1561010165","doi":"10.1109/snpd-sawn.2005.4","title":"A Method Level Based Approach for OO Integration Testing: An Experimental Study","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Integration testing; Java; Dependency (UML); Class (philosophy); Task (project management); Object-oriented programming; Software testing; Object (grammar); Software; Distributed computing; Software engineering; Programming language; Artificial intelligence; Systems engineering; Engineering","score_opus":0.21674407275810278,"score_gpt":0.39720876769489744,"score_spread":0.18046469493679465,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1561010165","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.976648,0.000099763194,0.021189542,0.00005056047,0.000020686588,0.0002788304,0.0000951525,0.0002928528,0.0013247399],"genre_scores_gemma":[0.9545596,0.000114863156,0.042790547,0.000051389678,0.00002886565,0.0006616355,0.00030257882,0.00009861461,0.0013917537],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967289,0.0015885538,0.00028611996,0.00045027377,0.00074117334,0.00020495633],"domain_scores_gemma":[0.9805546,0.013719919,0.0012503278,0.0028934898,0.0009872281,0.00059441],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033837964,0.0007173277,0.0005100792,0.0007650626,0.00042472626,0.0006125981,0.0012359415,0.001034905,0.0026376483],"category_scores_gemma":[0.012335548,0.00032268988,0.00041014163,0.00056161766,0.0008065274,0.0012755917,0.00090227555,0.0009946328,0.00039490432],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009723979,0.081251495,0.03127578,0.0017463437,0.00039124675,0.001043041,0.004528025,0.023237675,0.5070873,0.0067898813,0.0019620052,0.33096325],"study_design_scores_gemma":[0.0046057166,0.16059257,0.091476224,0.0002055379,0.0006617927,0.0021944353,0.002151751,0.313538,0.4060136,0.00593305,0.012368546,0.0002587555],"about_ca_topic_score_codex":0.00067331886,"about_ca_topic_score_gemma":0.0005311972,"teacher_disagreement_score":0.0033837964,"about_ca_system_score_codex":0.00035382446,"about_ca_system_score_gemma":0.00033036494,"threshold_uncertainty_score":0.01789546},"labels":[],"label_agreement":null},{"id":"W1564813932","doi":"10.1109/icnp.1993.340889","title":"A framework for interoperability testing of network protocols","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Interoperability; Computer science; Programming language; Protocol (science); Implementation; Formal specification; Relation (database); Interpretation (philosophy); Formal methods; Expressive power; Theoretical computer science; Data mining; World Wide Web","score_opus":0.15236436299207198,"score_gpt":0.3469682107810886,"score_spread":0.1946038477890166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1564813932","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008848368,0.00033897042,0.9919551,0.00070568244,0.00009641194,0.00014914371,0.00006499321,0.00054395146,0.0052607926],"genre_scores_gemma":[0.06500478,0.00083899964,0.9277679,0.0006594543,0.00023262942,0.0009921457,0.00041814518,0.00020923265,0.0038767634],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9891016,0.0052316003,0.0009455423,0.0010571056,0.002969671,0.00069453503],"domain_scores_gemma":[0.99032825,0.005447358,0.0005318722,0.002016778,0.0012900704,0.00038562034],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010412426,0.0023357125,0.0014978987,0.0043834476,0.0020021477,0.005848632,0.0053265155,0.0035394197,0.0051223584],"category_scores_gemma":[0.01478818,0.0014848037,0.004446773,0.0029459216,0.009757335,0.0089591835,0.0045101326,0.0069684032,0.0013178481],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000111040235,0.000037068807,0.000113548725,0.00008266654,0.000020621294,0.00016593703,0.00022825527,0.009983758,0.00056323974,0.9725018,0.0013570194,0.014934947],"study_design_scores_gemma":[0.000042630338,0.00006218847,0.00008657604,0.00021449407,0.000038967504,0.00024164401,0.00012169875,0.08073671,0.0013608586,0.875631,0.04142501,0.000038115097],"about_ca_topic_score_codex":0.0069526625,"about_ca_topic_score_gemma":0.0041113915,"teacher_disagreement_score":0.010412426,"about_ca_system_score_codex":0.003927085,"about_ca_system_score_gemma":0.0044970536,"threshold_uncertainty_score":0.055066824},"labels":[],"label_agreement":null},{"id":"W1564936289","doi":"","title":"Model based test suite minimization using metaheuristics","year":2011,"lang":"en","type":"article","venue":"Australasian Journal of Paramedicine","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Edith Cowan University; Royal Academy of Engineering; Carleton University","keywords":"Minification; Metaheuristic; Test suite; Computer science; Suite; Test (biology); Algorithm; Test case; Machine learning; Geology; Geography; Programming language","score_opus":0.09792129475973263,"score_gpt":0.3059081618920331,"score_spread":0.2079868671323005,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1564936289","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044919003,0.0014382341,0.94203186,0.000509793,0.00010901174,0.0003602381,0.00025866853,0.0010873602,0.009285893],"genre_scores_gemma":[0.40929452,0.0008718096,0.5796635,0.00045299638,0.00007266317,0.001271309,0.0010957768,0.00048726788,0.0067901365],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990004,0.00043732434,0.000046856007,0.00015158576,0.00020678139,0.00015723734],"domain_scores_gemma":[0.99576473,0.0034112483,0.00023611783,0.00012018752,0.00035556973,0.00011215066],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017563935,0.0019680087,0.0017913977,0.0025190124,0.00051552703,0.0014055336,0.0018917251,0.0016979842,0.0046383794],"category_scores_gemma":[0.0053532296,0.0009924751,0.0023772449,0.0015357243,0.0008535734,0.000835401,0.0014732336,0.0017861981,0.00054791203],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057869816,0.0000846488,0.0005935597,0.00008523568,0.00006675955,0.00005449785,0.000025071291,0.9633238,0.0006651549,0.004018608,0.000818181,0.030206662],"study_design_scores_gemma":[0.000014761759,0.000034224293,0.00008127676,0.000015362162,0.000015719314,0.000012519985,0.000011949285,0.9964813,0.00017242756,0.0028037073,0.00035353497,0.000003162992],"about_ca_topic_score_codex":0.0062677665,"about_ca_topic_score_gemma":0.0062595657,"teacher_disagreement_score":0.0062677665,"about_ca_system_score_codex":0.0019698462,"about_ca_system_score_gemma":0.002154893,"threshold_uncertainty_score":0.015516877},"labels":[],"label_agreement":null},{"id":"W1568097335","doi":"10.1007/978-1-4471-0719-4_31","title":"Using UML to Partially Automate Generation of Scenario-Based Test Drivers","year":2001,"lang":"en","type":"book-chapter","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Unified Modeling Language; Schedule; Computer science; Software engineering; Task (project management); Software development; Software; Systems engineering; Engineering; Programming language; Operating system","score_opus":0.10470659731656258,"score_gpt":0.29566436622660186,"score_spread":0.1909577689100393,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1568097335","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016918844,0.00006264882,0.9613233,0.000104577484,0.00003272909,0.000138655,0.00022228225,0.018461555,0.0027353833],"genre_scores_gemma":[0.21832271,0.00011070236,0.775196,0.000101838494,0.000011604701,0.00024134743,0.0010341496,0.0028159318,0.002165624],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988422,0.0004369666,0.000097170894,0.00016203345,0.00036715833,0.00009447164],"domain_scores_gemma":[0.99393326,0.0038316099,0.0004022033,0.0010775556,0.0006503749,0.000104892286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014648415,0.001050706,0.0004067202,0.0013372909,0.00032108618,0.0015748498,0.0012351549,0.0008977674,0.0043388107],"category_scores_gemma":[0.011297796,0.0011226059,0.00080763863,0.00046913253,0.0005946842,0.0018817623,0.0014156948,0.001321247,0.0014259062],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040773692,0.00038640332,0.006346691,0.00056584645,0.00016100744,0.0011563475,0.0018930396,0.16368337,0.0689594,0.060230628,0.014897684,0.6813118],"study_design_scores_gemma":[0.000089380694,0.00010732385,0.0008160884,0.0001740044,0.0000844258,0.00068246294,0.00012462055,0.8565431,0.08289006,0.032857783,0.025566828,0.00006389139],"about_ca_topic_score_codex":0.0024421946,"about_ca_topic_score_gemma":0.0032857882,"teacher_disagreement_score":0.0043388107,"about_ca_system_score_codex":0.0006368725,"about_ca_system_score_gemma":0.00103993,"threshold_uncertainty_score":0.014514744},"labels":[],"label_agreement":null},{"id":"W1568757735","doi":"10.1007/978-3-642-12200-2_32","title":"Finding the Best CAFE Is NP-Hard","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Clique; Binary number; Computer science; Reduction (mathematics); Time complexity; Enhanced Data Rates for GSM Evolution; Computational complexity theory; Polynomial; Hardness of approximation; Theoretical computer science; Discrete mathematics; Combinatorics; Algorithm; Approximation algorithm; Mathematics; Arithmetic; Artificial intelligence","score_opus":0.03334229128567038,"score_gpt":0.2729033384576659,"score_spread":0.23956104717199553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1568757735","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19798718,0.006891035,0.54884803,0.022939669,0.0010042919,0.00051564706,0.0065258755,0.012701029,0.20258717],"genre_scores_gemma":[0.4254665,0.0021970482,0.51530933,0.001298106,0.00037756152,0.00025991802,0.0037592475,0.0026726576,0.048659593],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9984458,0.0002495249,0.00009615226,0.00057686184,0.00031774386,0.00031382704],"domain_scores_gemma":[0.9929046,0.004829609,0.00031258786,0.0011783269,0.0005229692,0.00025189944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001135371,0.0015984388,0.0026152434,0.0012355684,0.003123313,0.0043120566,0.0021462261,0.0035255789,0.029222695],"category_scores_gemma":[0.008008618,0.0017119094,0.002644218,0.0025394089,0.0024812345,0.01465689,0.0024863759,0.0051291524,0.0052227285],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009288542,0.0006383796,0.0030938105,0.0023132036,0.00027555483,0.0006356777,0.00074495375,0.033835996,0.009761241,0.3119076,0.18734579,0.4485189],"study_design_scores_gemma":[0.00030044,0.00016323864,0.0009123042,0.0003089744,0.00025754696,0.0011087372,0.0006588283,0.06499899,0.007851515,0.8402494,0.08310343,0.000086581946],"about_ca_topic_score_codex":0.0033085758,"about_ca_topic_score_gemma":0.0063905967,"teacher_disagreement_score":0.029222695,"about_ca_system_score_codex":0.0017396532,"about_ca_system_score_gemma":0.0016655362,"threshold_uncertainty_score":0.097759664},"labels":[],"label_agreement":null},{"id":"W1570631586","doi":"10.1002/sec.1290","title":"An empirical investigation into path divergences for concolic execution using CREST","year":2015,"lang":"en","type":"article","venue":"Security and Communication Networks","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Core Research for Evolutional Science and Technology; National Natural Science Foundation of China","keywords":"Concolic testing; Computer science; Soundness; Path (computing); Test suite; Empirical research; Divergence (linguistics); Test case; Symbolic execution; Test (biology); Theoretical computer science; Software; Programming language; Machine learning; Mathematics; Statistics","score_opus":0.08112494552671053,"score_gpt":0.35457969414723567,"score_spread":0.27345474862052516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1570631586","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99165726,0.0001237719,0.0071366327,0.00008606206,0.0000082098395,0.000060946517,0.00022325864,0.00021342898,0.0004904447],"genre_scores_gemma":[0.9940315,0.00002877871,0.005256009,0.000019246558,0.000004333464,0.000035886937,0.0004963017,0.000043027045,0.00008489906],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9812262,0.0072499504,0.0016026555,0.0026240458,0.0065705176,0.000726612],"domain_scores_gemma":[0.65207773,0.2790455,0.027059615,0.020732632,0.018673012,0.0024115937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011937064,0.0005008989,0.00039723923,0.0026862563,0.00066331984,0.0010281556,0.001419954,0.00089848746,0.0007583061],"category_scores_gemma":[0.1389539,0.0003520442,0.00043578463,0.0023421862,0.0016319567,0.001987237,0.0012953893,0.0019092015,0.00016793273],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012177629,0.0012386811,0.8316823,0.00051573484,0.00030633117,0.0012148619,0.0036479486,0.023976037,0.00956052,0.004740753,0.003229905,0.118669175],"study_design_scores_gemma":[0.00016722012,0.0028566006,0.5392368,0.00023346856,0.00027807563,0.003982604,0.0038312783,0.41070873,0.02308359,0.008544422,0.0069051827,0.00017205486],"about_ca_topic_score_codex":0.0014671303,"about_ca_topic_score_gemma":0.0024354928,"teacher_disagreement_score":0.011937064,"about_ca_system_score_codex":0.0008805362,"about_ca_system_score_gemma":0.0008741865,"threshold_uncertainty_score":0.06313002},"labels":[],"label_agreement":null},{"id":"W1571486493","doi":"10.1109/ccece.2000.849660","title":"The architecture of a Java coverage tool","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Java; Instrumentation (computer programming); Computer science; Flexibility (engineering); Architecture; Domain (mathematical analysis); Software engineering; Operating system; Embedded system; Computer architecture","score_opus":0.015406256352965074,"score_gpt":0.2183285589213015,"score_spread":0.20292230256833643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1571486493","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00503165,0.00028347806,0.9066417,0.00023961475,0.00006179864,0.0002169302,0.00036514873,0.0799236,0.007236012],"genre_scores_gemma":[0.13939846,0.0004862326,0.8363185,0.00065514824,0.00013205293,0.00053592253,0.0035263966,0.009806857,0.009140468],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9975193,0.00026708236,0.00021191196,0.00041317617,0.0013925977,0.0001959002],"domain_scores_gemma":[0.9966594,0.0012071832,0.00029380203,0.00054203626,0.001048921,0.00024871883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018165618,0.00065065746,0.0007630912,0.0033384236,0.0005505355,0.002816111,0.0023172726,0.0011142589,0.006717155],"category_scores_gemma":[0.00743387,0.0009470666,0.0007971043,0.001483885,0.00067759544,0.003124902,0.0017528163,0.0014612658,0.004083415],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006482181,0.00043115043,0.010160209,0.00080810266,0.00015482726,0.0009208057,0.0009784359,0.019435652,0.05646204,0.04453251,0.043586336,0.8218817],"study_design_scores_gemma":[0.0003512808,0.0005665852,0.010174957,0.0006052599,0.00031737937,0.0042661186,0.00032456513,0.45950544,0.088690884,0.065923594,0.36889884,0.00037511156],"about_ca_topic_score_codex":0.003177479,"about_ca_topic_score_gemma":0.0017723199,"teacher_disagreement_score":0.006717155,"about_ca_system_score_codex":0.00061897736,"about_ca_system_score_gemma":0.0012948853,"threshold_uncertainty_score":0.02247107},"labels":[],"label_agreement":null},{"id":"W1576774211","doi":"10.1007/978-3-642-01187-0_26","title":"Integration Testing of Web Applications and Databases Using TTCN-3","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in business information processing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Variety (cybernetics); Java; Integration testing; SQL; Web application; Database; Software engineering; Programming language; Data integration; World Wide Web; Software; Artificial intelligence","score_opus":0.04183812627414476,"score_gpt":0.2798233376161275,"score_spread":0.23798521134198272,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1576774211","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17721477,0.0005642495,0.77324617,0.00033994784,0.00014348899,0.00036688574,0.0007070208,0.023453293,0.023964224],"genre_scores_gemma":[0.620111,0.00018414414,0.36034158,0.00019778931,0.000031377982,0.00020712335,0.002118424,0.0017082209,0.015100363],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99570775,0.0010438355,0.00030496818,0.0004962984,0.002065937,0.00038118463],"domain_scores_gemma":[0.9942281,0.0024568406,0.00032236218,0.0012374092,0.0015678895,0.00018728484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002396798,0.0008496394,0.000583363,0.0016709738,0.0004623366,0.0013519189,0.0021370659,0.0011189955,0.0073446687],"category_scores_gemma":[0.0067141145,0.00046036206,0.0008699943,0.0012201646,0.0007093827,0.0017658451,0.0013381693,0.0009111249,0.0011000464],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014638145,0.0011305885,0.020375492,0.00067045266,0.00021897121,0.0015090307,0.00078172813,0.103436954,0.121779904,0.030221457,0.015548856,0.7028627],"study_design_scores_gemma":[0.00017529154,0.00072433037,0.0064691436,0.00018834145,0.0001621635,0.0013525384,0.00018323225,0.7464635,0.20149475,0.019057747,0.023644622,0.00008433439],"about_ca_topic_score_codex":0.0070983307,"about_ca_topic_score_gemma":0.007877187,"teacher_disagreement_score":0.0073446687,"about_ca_system_score_codex":0.0009176247,"about_ca_system_score_gemma":0.0015580292,"threshold_uncertainty_score":0.024570346},"labels":[],"label_agreement":null},{"id":"W1576968470","doi":"10.1007/978-3-642-16901-4_14","title":"API Conformance Verification for Java Programs","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Programming language; Executable; Java; Control flow; Operating system; Concurrency; Conformance testing; Application programming interface; Software engineering","score_opus":0.02692841412498071,"score_gpt":0.2677013130703781,"score_spread":0.24077289894539738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1576968470","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024446469,0.00051134627,0.96036077,0.00020648398,0.00010332684,0.00012116754,0.00009971952,0.008208456,0.0059423363],"genre_scores_gemma":[0.60158354,0.00076845096,0.38312536,0.0002678202,0.00011361524,0.00019050122,0.00072033936,0.0031685107,0.010061871],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9951207,0.0010572035,0.0003339154,0.0006127887,0.0024432582,0.0004320836],"domain_scores_gemma":[0.99297357,0.0034278727,0.00046002324,0.0018345175,0.0012307069,0.0000733425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002823163,0.0008711021,0.00085859746,0.0013021481,0.0007749064,0.0014733942,0.0019867793,0.0011283632,0.004593685],"category_scores_gemma":[0.011043658,0.00095103297,0.0018852026,0.00078116095,0.001329308,0.0024650253,0.0015407673,0.0026580605,0.0011767698],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006349717,0.0003028476,0.0035681243,0.0009153818,0.00017876242,0.0010320457,0.0009318325,0.04559821,0.0942329,0.10188306,0.00955407,0.7411678],"study_design_scores_gemma":[0.00018256961,0.000457389,0.0037549753,0.0004811688,0.0002777643,0.0018050242,0.00023949805,0.5282024,0.225104,0.21233171,0.027000071,0.00016334622],"about_ca_topic_score_codex":0.0024584408,"about_ca_topic_score_gemma":0.0022387933,"teacher_disagreement_score":0.004593685,"about_ca_system_score_codex":0.0007535564,"about_ca_system_score_gemma":0.001100003,"threshold_uncertainty_score":0.015367329},"labels":[],"label_agreement":null},{"id":"W1577013748","doi":"10.1109/pacrim.2001.953527","title":"Routing reliability analysis of partially disjoint paths","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Computer network; Static routing; Equal-cost multi-path routing; Open Shortest Path First; Link-state routing protocol; Routing protocol; Dynamic Source Routing; Routing Information Protocol; Zone Routing Protocol; Multipath routing; Enhanced Interior Gateway Routing Protocol; Reliability (semiconductor); Distributed computing; Path vector protocol; Routing (electronic design automation)","score_opus":0.0308666070486083,"score_gpt":0.25693357631026786,"score_spread":0.22606696926165956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1577013748","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22013299,0.0014799741,0.7707288,0.00028723458,0.000037426653,0.000041746513,0.00011132924,0.00018582064,0.00699461],"genre_scores_gemma":[0.9744995,0.00079780177,0.023106877,0.00002922999,0.000036667865,0.000055572655,0.000085276486,0.0000402363,0.0013487296],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991799,0.0002504104,0.000030135416,0.00009119925,0.0003637239,0.00008456596],"domain_scores_gemma":[0.99617505,0.002215008,0.00043313854,0.00031434718,0.0007800517,0.00008244344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011198065,0.000485734,0.0004186005,0.0010375471,0.00032767144,0.00047052608,0.0005025747,0.00037079694,0.0009472119],"category_scores_gemma":[0.0069094375,0.00032699236,0.0005026581,0.0004776653,0.00060930406,0.00096918945,0.00064209773,0.00046731066,0.00015192333],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007047052,0.000014240536,0.0018088883,0.00008584728,0.00004174743,0.00022941976,0.0001403385,0.89124495,0.007463056,0.079269834,0.00067530427,0.018955901],"study_design_scores_gemma":[0.0000029549913,0.000022865643,0.00047618724,0.000007717401,0.000010406659,0.000093699695,0.000018727664,0.98098457,0.0008653843,0.017003918,0.0005071978,0.0000062892814],"about_ca_topic_score_codex":0.001730213,"about_ca_topic_score_gemma":0.00067764585,"teacher_disagreement_score":0.001730213,"about_ca_system_score_codex":0.00068819674,"about_ca_system_score_gemma":0.00047648157,"threshold_uncertainty_score":0.0059221983},"labels":[],"label_agreement":null},{"id":"W1577404745","doi":"10.1007/s10009-007-0044-z","title":"The software model checker Blast","year":2007,"lang":"en","type":"article","venue":"International Journal on Software Tools for Technology Transfer","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":594,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Predicate abstraction; Predicate (mathematical logic); Programming language; Memory safety; Model checking; Programmer; Program analysis; Symbolic execution; Static analysis; Software","score_opus":0.029266124064769698,"score_gpt":0.3019749021356067,"score_spread":0.272708778070837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1577404745","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030684728,0.000747805,0.85916394,0.0012193903,0.0004702523,0.00013640468,0.0011014462,0.075376466,0.031099547],"genre_scores_gemma":[0.43899733,0.0008296994,0.52607524,0.0011386661,0.00017714707,0.00019560504,0.0033601222,0.00766588,0.02156033],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99801517,0.0005311157,0.00011034089,0.0003188596,0.000876713,0.00014789692],"domain_scores_gemma":[0.9967263,0.0012889211,0.00016495664,0.0011925183,0.0005339804,0.000093411494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015388378,0.0009394044,0.0010077625,0.0013998662,0.0007671447,0.0018744217,0.0017273582,0.0015231547,0.009388052],"category_scores_gemma":[0.009137308,0.000836429,0.00095891743,0.0010229489,0.0011384494,0.0036142801,0.0020334453,0.0022984454,0.0034627686],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010958522,0.0003966122,0.004156827,0.0008025023,0.00026012483,0.0009579579,0.0003163709,0.10888505,0.029580772,0.27376372,0.097338066,0.48244604],"study_design_scores_gemma":[0.0004143246,0.00023535594,0.00079142425,0.00027538367,0.00021956506,0.000743638,0.000073722185,0.6032544,0.04506527,0.2504209,0.09844049,0.00006556918],"about_ca_topic_score_codex":0.0026645786,"about_ca_topic_score_gemma":0.0036865687,"teacher_disagreement_score":0.009388052,"about_ca_system_score_codex":0.00059171795,"about_ca_system_score_gemma":0.002096661,"threshold_uncertainty_score":0.031406164},"labels":[],"label_agreement":null},{"id":"W1581911077","doi":"10.1007/978-3-540-68255-4_33","title":"Build Notifications in Agile Environments","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in business information processing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Codebase; Agile software development; Computer science; Software engineering; Suite; Software; Timeout; Software development; Operating system","score_opus":0.017591760877402635,"score_gpt":0.23294724924668386,"score_spread":0.21535548836928123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1581911077","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054338362,0.0009942917,0.9038315,0.00074892526,0.0002838029,0.00009358737,0.000072066316,0.00874832,0.030889167],"genre_scores_gemma":[0.51172805,0.0011539306,0.4520588,0.00018735502,0.00008644782,0.00016201471,0.00027699335,0.0019684967,0.032377988],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99769884,0.0007390889,0.00013131122,0.00022641724,0.0009793635,0.00022505711],"domain_scores_gemma":[0.9924601,0.004108177,0.000498529,0.0018729577,0.0005091491,0.00055104203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003817445,0.0005950826,0.0003861339,0.0008234528,0.0008820565,0.0026737498,0.001498403,0.0012731937,0.0045817997],"category_scores_gemma":[0.01223116,0.0010800883,0.00036868875,0.00090026774,0.00097502844,0.004428616,0.0025233713,0.002874784,0.0012083673],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005093809,0.00037268453,0.0052691596,0.00037148673,0.000043853517,0.0011452723,0.006110274,0.021782523,0.01373104,0.19211066,0.014708442,0.7438453],"study_design_scores_gemma":[0.0002853553,0.0007440073,0.006450657,0.00073565013,0.00018435277,0.0035830813,0.0037040745,0.34071738,0.052625246,0.30452347,0.2862045,0.0002422807],"about_ca_topic_score_codex":0.0014469188,"about_ca_topic_score_gemma":0.0022440336,"teacher_disagreement_score":0.0045817997,"about_ca_system_score_codex":0.00046849798,"about_ca_system_score_gemma":0.0007624559,"threshold_uncertainty_score":0.020188808},"labels":[],"label_agreement":null},{"id":"W1582381037","doi":"10.1007/978-3-540-68255-4_27","title":"Multi-modal Functional Test Execution","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in business information processing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Debugging; Computer science; Modal; Layer (electronics); Programming language; Software engineering; Test case; Overhead (engineering); Embedded system","score_opus":0.02655807541538957,"score_gpt":0.23848792422266754,"score_spread":0.21192984880727797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1582381037","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030478211,0.00015497512,0.94590145,0.000177156,0.000062684376,0.00009201677,0.00022767899,0.008874805,0.014030998],"genre_scores_gemma":[0.6526618,0.00011622309,0.33129206,0.00022939123,0.000034255794,0.00016807464,0.00077349396,0.0011625608,0.013562036],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99833184,0.00042465833,0.00010155938,0.0002988001,0.000588475,0.00025473096],"domain_scores_gemma":[0.9964275,0.0018178151,0.00019259095,0.0008953916,0.0005633523,0.00010341037],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018788644,0.0010498326,0.00044803278,0.0011491916,0.0004802819,0.00089366746,0.0018110451,0.0009375745,0.013557437],"category_scores_gemma":[0.0043877764,0.00038545224,0.0007786805,0.00053049915,0.0010490833,0.0023621023,0.0017374546,0.0010361737,0.0016176422],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016357919,0.0005158914,0.0039281757,0.000643773,0.00011076795,0.0010281617,0.0006753037,0.060365684,0.06432026,0.15554342,0.01227267,0.69896007],"study_design_scores_gemma":[0.0001204904,0.00054231304,0.0022674263,0.00024279475,0.00012151093,0.00133388,0.00020936901,0.68605,0.11076204,0.17199889,0.026264174,0.0000871003],"about_ca_topic_score_codex":0.0013054954,"about_ca_topic_score_gemma":0.0023284182,"teacher_disagreement_score":0.013557437,"about_ca_system_score_codex":0.00053397473,"about_ca_system_score_gemma":0.0006701624,"threshold_uncertainty_score":0.045354128},"labels":[],"label_agreement":null},{"id":"W1582552606","doi":"","title":"Signature required making simulink data flow and interfaces explicit","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Signature (topology); Computer science; Reuse; Generalization; Interface (matter); Comprehension; Automotive industry; Software engineering; Programming language; Embedded system; Operating system; Engineering","score_opus":0.05427294548107347,"score_gpt":0.30487206208818773,"score_spread":0.25059911660711426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1582552606","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061763342,0.000044939472,0.9294625,0.0001913261,0.00004074749,0.00009307884,0.000107280415,0.0052956818,0.0030011323],"genre_scores_gemma":[0.61443037,0.000103562816,0.3807973,0.00013469627,0.000022247077,0.00010298753,0.00029802832,0.0011260033,0.002984763],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99782866,0.0006826767,0.00025994147,0.0003098538,0.0007218808,0.00019703228],"domain_scores_gemma":[0.9883223,0.004943077,0.0013390125,0.0033063882,0.0019171626,0.00017205652],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021670875,0.0005832946,0.0004363394,0.0006397811,0.00058192026,0.0016504609,0.0010137134,0.0009802426,0.0034064353],"category_scores_gemma":[0.012136108,0.0005253457,0.0004902203,0.0006004139,0.0015332315,0.0037389512,0.0018253236,0.0017180035,0.00074336736],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014063352,0.00043381154,0.009984848,0.0008678357,0.00009485962,0.0019340743,0.0050394265,0.12189231,0.2077951,0.27791503,0.004327703,0.36830863],"study_design_scores_gemma":[0.000080649355,0.00020463803,0.0006573899,0.00017502408,0.00010628358,0.00069821934,0.00036279354,0.31356367,0.5831581,0.065570794,0.03535221,0.00007017517],"about_ca_topic_score_codex":0.0012654044,"about_ca_topic_score_gemma":0.0013444992,"teacher_disagreement_score":0.0034064353,"about_ca_system_score_codex":0.0007361002,"about_ca_system_score_gemma":0.0018898867,"threshold_uncertainty_score":0.011460781},"labels":[],"label_agreement":null},{"id":"W1582795712","doi":"10.1007/b95400","title":"Formal approaches to software testing : third International Workshop on Formal Approaches to Testing of Software, FATES 2003, Montreal, Quebec, Canada, October 6th, 2003 : revised papers","year":2004,"lang":"en","type":"book","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"White-box testing; Computer science; Test Management Approach; Non-regression testing; Keyword-driven testing; Manual testing; Model-based testing; Programming language; Test script; System integration testing; Software engineering; Black-box testing; Regression testing; Test case; Code coverage; Test strategy; Conformance testing; Software reliability testing; Software testing; Software; Software system; Software development; Operating system; Software quality; Software construction; Machine learning","score_opus":0.10958301892259165,"score_gpt":0.23538909870121832,"score_spread":0.12580607977862668,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1582795712","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0057556243,0.2677196,0.5180172,0.025013307,0.017964609,0.0005067694,0.0030749298,0.005800736,0.15614726],"genre_scores_gemma":[0.051889442,0.14255601,0.12998703,0.002158007,0.0034074478,0.00033248487,0.0052640447,0.0029258612,0.6614797],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987166,0.00017086176,0.00007793961,0.00016323285,0.00073127187,0.0001401029],"domain_scores_gemma":[0.99676347,0.00096922455,0.0000963676,0.000250549,0.0016790795,0.0002412617],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002424904,0.0017434342,0.0015006422,0.0029873776,0.0010948313,0.0049363524,0.0022846784,0.0013033523,0.029427296],"category_scores_gemma":[0.004256587,0.0011614508,0.00080747745,0.004447476,0.0026139403,0.0030369083,0.0008459412,0.0020686148,0.0062250462],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000078371806,0.00008895939,0.0003579464,0.0008725344,0.00003028062,0.00010790079,0.00067535054,0.004104725,0.0025639855,0.045054663,0.5796781,0.3663872],"study_design_scores_gemma":[0.00008562041,0.00007785796,0.0019418196,0.00082105375,0.000056387187,0.0004377613,0.00027219896,0.009156832,0.0034762952,0.048897587,0.9347066,0.00006993474],"about_ca_topic_score_codex":0.20936504,"about_ca_topic_score_gemma":0.3089982,"teacher_disagreement_score":0.20936504,"about_ca_system_score_codex":0.008479737,"about_ca_system_score_gemma":0.008880382,"threshold_uncertainty_score":0.4162928},"labels":[],"label_agreement":null},{"id":"W1586015080","doi":"","title":"Qualitative observations from software code inspection experiments","year":2002,"lang":"en","type":"article","venue":"Conference of the Centre for Advanced Studies on Collaborative Research","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Software inspection; Computer science; Software; Code (set theory); Software engineering; Software quality; Programming language; Software development","score_opus":0.4511467688952683,"score_gpt":0.4903815943352252,"score_spread":0.03923482543995693,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1586015080","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95746547,0.00047467338,0.03469554,0.0006554042,0.00005767768,0.0004428115,0.0013483657,0.00062314566,0.0042368933],"genre_scores_gemma":[0.9897288,0.00013334992,0.007875596,0.0002329724,0.000020116822,0.00033799684,0.0006626325,0.00012666029,0.00088179484],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9785138,0.010907549,0.0011151107,0.0016935221,0.0069698584,0.0008001181],"domain_scores_gemma":[0.7274757,0.20297568,0.024805946,0.016010987,0.026147112,0.0025846525],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011820192,0.0007551609,0.00071998016,0.0021470003,0.0010799433,0.0010539277,0.0012186269,0.0012775237,0.0010615947],"category_scores_gemma":[0.12416905,0.00047014854,0.00045418492,0.0017413417,0.001938814,0.0013308846,0.0016300573,0.002063788,0.00044018877],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005287027,0.003261523,0.29610893,0.004881228,0.00070109055,0.0026856728,0.11216825,0.019883199,0.35897842,0.0059341374,0.016472558,0.17363794],"study_design_scores_gemma":[0.00059247157,0.011108673,0.6046214,0.0011205092,0.00047695177,0.0021215046,0.04897046,0.051659428,0.22799207,0.018872505,0.031632952,0.0008310519],"about_ca_topic_score_codex":0.001166308,"about_ca_topic_score_gemma":0.002107993,"teacher_disagreement_score":0.011820192,"about_ca_system_score_codex":0.0009752863,"about_ca_system_score_gemma":0.00056948,"threshold_uncertainty_score":0.06251192},"labels":[],"label_agreement":null},{"id":"W1590993614","doi":"10.1007/978-3-642-12566-9_13","title":"On Software Certification: We Need Product-Focused Approaches","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Certification; Software engineering; Product (mathematics); Software; Programming language; Management; Mathematics","score_opus":0.05594045579544545,"score_gpt":0.2496989663611623,"score_spread":0.19375851056571683,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1590993614","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034760318,0.014316144,0.8147446,0.03487379,0.0013839247,0.00007172459,0.000054031316,0.00068141054,0.13039836],"genre_scores_gemma":[0.2803537,0.04047457,0.55344546,0.017828748,0.0038095736,0.00034781732,0.000362394,0.0014208178,0.10195694],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.996705,0.001074846,0.00014830081,0.00036360638,0.0015123495,0.0001960459],"domain_scores_gemma":[0.98689044,0.006418233,0.00041014055,0.0027146123,0.0030904328,0.0004761967],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005333131,0.0013937359,0.0011934248,0.0023519201,0.0012992938,0.00473378,0.002525601,0.0039262627,0.012294756],"category_scores_gemma":[0.010349727,0.001017613,0.0010243967,0.001848748,0.007813109,0.028378462,0.0033972678,0.009295151,0.0036693122],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000109550965,0.00005789048,0.00019550545,0.00025229063,0.000015076465,0.00008400962,0.00035641287,0.0015850288,0.0005283129,0.89441013,0.0145972725,0.08790712],"study_design_scores_gemma":[0.000006276933,0.000026954322,0.0001289566,0.00026581023,0.000011991048,0.00016282346,0.00015732912,0.0041393405,0.00043247157,0.92681885,0.0678327,0.00001642518],"about_ca_topic_score_codex":0.0011238385,"about_ca_topic_score_gemma":0.0015767862,"teacher_disagreement_score":0.012294756,"about_ca_system_score_codex":0.001794944,"about_ca_system_score_gemma":0.0016678132,"threshold_uncertainty_score":0.041130126},"labels":[],"label_agreement":null},{"id":"W1591778339","doi":"10.33915/etd.1237","title":"Random search of AND-OR graphs representing finite-state models","year":2002,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Finite-state machine; Testability; Theoretical computer science; Model checking; State (computer science); Random graph; Graph; Algorithm; Mathematics","score_opus":0.06222283750590606,"score_gpt":0.31411476337761907,"score_spread":0.251891925871713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1591778339","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7105261,0.00060300785,0.27742493,0.0006528761,0.00003797532,0.00025189348,0.0016158168,0.002696189,0.0061911778],"genre_scores_gemma":[0.84364986,0.00022598024,0.1507966,0.00014894689,0.0000104384335,0.00031941864,0.0031591128,0.0003106276,0.0013788854],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982039,0.0009727244,0.000080521844,0.000305128,0.00027578222,0.00016184298],"domain_scores_gemma":[0.9788824,0.017683031,0.0008804237,0.0016476277,0.0007466403,0.00015988677],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002485437,0.0007507757,0.00087742816,0.0017861233,0.00068841025,0.000918625,0.0011544895,0.0010038031,0.002688153],"category_scores_gemma":[0.01608486,0.0005378295,0.0015232525,0.0014245211,0.0011055524,0.0019527948,0.00080411305,0.0011589216,0.00027130204],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072863477,0.00022827498,0.005094373,0.00037229684,0.00016418275,0.00028777294,0.00033196554,0.907489,0.007059174,0.045941956,0.002513983,0.02978828],"study_design_scores_gemma":[0.00007737685,0.000082748455,0.00050475035,0.000024247325,0.00005114283,0.00005132836,0.000088138375,0.970044,0.0032039867,0.025060158,0.000798028,0.0000140469465],"about_ca_topic_score_codex":0.004631092,"about_ca_topic_score_gemma":0.010881307,"teacher_disagreement_score":0.004631092,"about_ca_system_score_codex":0.001912207,"about_ca_system_score_gemma":0.001492268,"threshold_uncertainty_score":0.013874114},"labels":[],"label_agreement":null},{"id":"W1593362647","doi":"10.1007/978-3-642-01187-0_12","title":"Model-Based Penetration Test Framework for Web Applications Using TTCN-3","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in business information processing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Penetration (warfare); Engineering; Operations research","score_opus":0.03229845071128463,"score_gpt":0.2863652810987781,"score_spread":0.2540668303874935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1593362647","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037607928,0.00010155123,0.97784084,0.00009656321,0.000023621715,0.00017018101,0.00015581906,0.01518149,0.0026691789],"genre_scores_gemma":[0.2328856,0.00035184808,0.75479466,0.00017521335,0.000031832984,0.00059981685,0.0014476549,0.0039033394,0.0058100163],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99779487,0.0006458182,0.00020114989,0.0002385941,0.00091460894,0.0002050652],"domain_scores_gemma":[0.99766845,0.0009327362,0.00016253693,0.0005683347,0.0005782131,0.000089589426],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024421085,0.0015522697,0.0011925356,0.0019111098,0.00065662095,0.0025344351,0.0031082325,0.0016052029,0.009059788],"category_scores_gemma":[0.0052993502,0.0008632874,0.0022262724,0.0008967526,0.0010426755,0.0027331652,0.00214386,0.0020257777,0.0016833304],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038303333,0.0005568294,0.004074168,0.0007450376,0.00029500565,0.0011401554,0.00063748664,0.44290522,0.018573565,0.18543114,0.019992383,0.32526597],"study_design_scores_gemma":[0.000034062723,0.00005907678,0.0002505629,0.00009473931,0.000052277624,0.00027132957,0.00004067617,0.9493134,0.006405311,0.029620362,0.013820609,0.00003750618],"about_ca_topic_score_codex":0.010293012,"about_ca_topic_score_gemma":0.01040009,"teacher_disagreement_score":0.010293012,"about_ca_system_score_codex":0.0012126537,"about_ca_system_score_gemma":0.002194115,"threshold_uncertainty_score":0.030308008},"labels":[],"label_agreement":null},{"id":"W1593942753","doi":"10.1007/978-3-642-16612-9_15","title":"Clara: A Framework for Partially Evaluating Finite-State Runtime Monitors Ahead of Time","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Waterloo","funders":"","keywords":"Computer science; Runtime verification; AspectJ; Compile time; Compiler; Residual; Static analysis; State (computer science); Runtime system; Property (philosophy); Programming language; Distributed computing; Aspect-oriented programming; Formal verification; Software; Algorithm","score_opus":0.034323428866062475,"score_gpt":0.3122384992071882,"score_spread":0.2779150703411257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1593942753","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018187603,0.00027650042,0.92629665,0.000098186174,0.000105258245,0.000091903006,0.00028736197,0.069302596,0.0017228175],"genre_scores_gemma":[0.079304166,0.000272128,0.9026127,0.00023197604,0.00015012076,0.00032486118,0.0010349673,0.011655738,0.004413326],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9936178,0.0016537253,0.00053252827,0.0013241641,0.0022843704,0.00058743847],"domain_scores_gemma":[0.9878181,0.006296372,0.0008797426,0.0032732133,0.001337213,0.00039528002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060761115,0.0033824826,0.0027289118,0.002335398,0.0013473044,0.005276554,0.008252964,0.0029149225,0.018549997],"category_scores_gemma":[0.01941569,0.0039264397,0.003742582,0.0014915976,0.002784384,0.008338798,0.0042709424,0.005356034,0.0056493366],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002824418,0.00035374705,0.0033924547,0.0022115116,0.0005482401,0.00070880895,0.0008782167,0.11622047,0.037632335,0.17535704,0.060111266,0.5997615],"study_design_scores_gemma":[0.00044368513,0.00027203685,0.00054819795,0.0002936956,0.0003618995,0.0004578771,0.000090468835,0.7648379,0.05481756,0.110288836,0.06733685,0.00025102854],"about_ca_topic_score_codex":0.006154794,"about_ca_topic_score_gemma":0.008816345,"teacher_disagreement_score":0.018549997,"about_ca_system_score_codex":0.002127454,"about_ca_system_score_gemma":0.0039740494,"threshold_uncertainty_score":0.062055945},"labels":[],"label_agreement":null},{"id":"W1594634494","doi":"10.1007/978-3-642-04468-7_19","title":"Towards Model-Based Automatic Testing of Attack Scenarios","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science","score_opus":0.043578089789563096,"score_gpt":0.28380566725869855,"score_spread":0.24022757746913545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1594634494","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0076409094,0.0001133475,0.98380697,0.00008513058,0.000022691542,0.00007962423,0.00009647396,0.006898087,0.0012567058],"genre_scores_gemma":[0.19868837,0.0001796068,0.79796636,0.00012377225,0.000029773717,0.00013220408,0.000607888,0.0008484853,0.0014235309],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965796,0.0010999996,0.00021603957,0.0004834744,0.0013702477,0.00025068154],"domain_scores_gemma":[0.9898003,0.005958881,0.00061386597,0.002273312,0.0011864417,0.00016721134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020417648,0.0015334885,0.0012107498,0.0022709656,0.00043897732,0.0022837438,0.0028117038,0.0018481119,0.0036734166],"category_scores_gemma":[0.0112462845,0.0012992023,0.0018290848,0.0010194756,0.0011884595,0.003609633,0.0025932589,0.0024470738,0.0016668974],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047623247,0.000499154,0.0034917304,0.0007201684,0.00027193868,0.0006080821,0.00043284404,0.36896867,0.059932265,0.059977103,0.012306421,0.49231532],"study_design_scores_gemma":[0.000030520663,0.0000450901,0.00020425644,0.000040069714,0.000033984805,0.0001647356,0.000025806981,0.9538096,0.009988752,0.03369315,0.001948816,0.000015190048],"about_ca_topic_score_codex":0.0021348738,"about_ca_topic_score_gemma":0.0035847877,"teacher_disagreement_score":0.0036734166,"about_ca_system_score_codex":0.00089661207,"about_ca_system_score_gemma":0.0010577585,"threshold_uncertainty_score":0.012288749},"labels":[],"label_agreement":null},{"id":"W1595781181","doi":"10.1007/978-3-642-02818-2_21","title":"Script InSight: Using Models to Explore JavaScript Code from the Browser View","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"JavaScript; Computer science; Unobtrusive JavaScript; Java; Rich Internet application; Tracing; World Wide Web; Scripting language; Context (archaeology); Programming language; Web application; Source code; Code (set theory); Software engineering","score_opus":0.11687458812616955,"score_gpt":0.2886179972318393,"score_spread":0.17174340910566974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1595781181","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009729181,0.00023644482,0.9309135,0.00032786606,0.000068552006,0.00008145529,0.000793206,0.0386804,0.019169329],"genre_scores_gemma":[0.21877813,0.0010502021,0.7277928,0.0003110496,0.00004932322,0.00037717077,0.0029038158,0.02280953,0.025928011],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999816,0.000042829277,0.0000103162965,0.000038683324,0.000069448004,0.00002270255],"domain_scores_gemma":[0.9991848,0.00047395518,0.000032564556,0.00015112625,0.000073493284,0.00008415486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003746647,0.0011151136,0.00047606617,0.000837714,0.0005204235,0.002653515,0.0015664353,0.001108057,0.022343643],"category_scores_gemma":[0.0025525896,0.00089910027,0.001145198,0.00055072864,0.00082892703,0.00489784,0.001901663,0.0019792242,0.004929097],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009847726,0.00030538594,0.0034758954,0.0010457364,0.00011429109,0.0012836448,0.0060724844,0.062246185,0.06056334,0.33315882,0.09316071,0.43758875],"study_design_scores_gemma":[0.00015055483,0.00012735561,0.0008832049,0.00035158833,0.00008279012,0.0006842158,0.00056584174,0.5284106,0.02798321,0.2396735,0.20095965,0.00012743629],"about_ca_topic_score_codex":0.0023535285,"about_ca_topic_score_gemma":0.0040800017,"teacher_disagreement_score":0.022343643,"about_ca_system_score_codex":0.00048876664,"about_ca_system_score_gemma":0.0005999108,"threshold_uncertainty_score":0.07474691},"labels":[],"label_agreement":null},{"id":"W1598641709","doi":"10.1007/978-3-540-73066-8_23","title":"An EFSM-Based Passive Fault Detection Approach","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Government of Ontario","keywords":"Computer science; Extended finite-state machine; Fault detection and isolation; Fault (geology); Finite-state machine; Artificial intelligence; Algorithm","score_opus":0.027082481015822873,"score_gpt":0.2738637657798862,"score_spread":0.24678128476406333,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1598641709","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008460193,0.00017912549,0.98598516,0.00008329291,0.000062271814,0.00004564201,0.00004065389,0.0016363133,0.003507331],"genre_scores_gemma":[0.4011794,0.00025668446,0.58786076,0.0002916346,0.00007539215,0.000101832855,0.00022806769,0.00021598225,0.0097902715],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992105,0.00015985053,0.000044204156,0.00015748503,0.00034688215,0.00008115941],"domain_scores_gemma":[0.9983664,0.000529454,0.00010391013,0.0004114753,0.00054803054,0.000040745734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008261923,0.00096671376,0.0008577789,0.0013179453,0.00048413582,0.0008297212,0.0022512947,0.0010827777,0.003560766],"category_scores_gemma":[0.0020211095,0.00035418253,0.0005789845,0.0005921318,0.00063475006,0.0018860203,0.0011611322,0.0009839596,0.0010343322],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004816421,0.0004401293,0.001506659,0.00043316363,0.000106472304,0.0003795214,0.00021452236,0.06450698,0.08724889,0.057296928,0.004902532,0.78248245],"study_design_scores_gemma":[0.000057521615,0.0003483757,0.0007612854,0.00008196399,0.0001324087,0.0006643913,0.000060421185,0.8831551,0.04523773,0.05772356,0.011737185,0.00004009663],"about_ca_topic_score_codex":0.0009517274,"about_ca_topic_score_gemma":0.0017876261,"teacher_disagreement_score":0.003560766,"about_ca_system_score_codex":0.00038142662,"about_ca_system_score_gemma":0.0007036113,"threshold_uncertainty_score":0.011911929},"labels":[],"label_agreement":null},{"id":"W1601589757","doi":"10.1007/978-3-540-24723-4_19","title":"Integrating the Soot Compiler Infrastructure into an IDE","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Compiler; Computer science; Debugging; Eclipse; Software engineering; Programming language; Byte; Development environment","score_opus":0.015372989434315964,"score_gpt":0.2634255679722774,"score_spread":0.24805257853796142,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1601589757","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006374908,0.00022406189,0.9512067,0.00016847576,0.00034053522,0.00010118031,0.00013278342,0.027051438,0.014399978],"genre_scores_gemma":[0.089490555,0.0005852708,0.87064636,0.00033180168,0.00012313749,0.00012300018,0.0010481562,0.012752101,0.024899608],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99871373,0.00018962703,0.000110763,0.00018664173,0.00060866686,0.00019070083],"domain_scores_gemma":[0.99773586,0.00074080215,0.00011042705,0.0007084998,0.00055199524,0.00015248425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018624292,0.0006593456,0.00070594193,0.00091664545,0.00048670382,0.003026052,0.0017236228,0.0007751135,0.0075830882],"category_scores_gemma":[0.0042338506,0.0009648201,0.0009387374,0.00054413103,0.0006034382,0.0033444725,0.0022041805,0.002352794,0.006020496],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006281284,0.00036739017,0.0028010583,0.000541095,0.00009528835,0.0007256394,0.00072547473,0.008949635,0.051239885,0.086487405,0.034943003,0.812496],"study_design_scores_gemma":[0.00031025757,0.00035438171,0.0018994653,0.00044297968,0.00023267465,0.0019189611,0.0002551046,0.14294682,0.14795332,0.09237025,0.61113864,0.0001772288],"about_ca_topic_score_codex":0.00074812566,"about_ca_topic_score_gemma":0.0013217649,"teacher_disagreement_score":0.0075830882,"about_ca_system_score_codex":0.00039501465,"about_ca_system_score_gemma":0.0016191939,"threshold_uncertainty_score":0.025367916},"labels":[],"label_agreement":null},{"id":"W1601919490","doi":"10.1007/978-3-540-76650-6_11","title":"Reducing Test Sequence Length Using Invertible Sequences","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Invertible matrix; Sequence (biology); Computer science; Finite-state machine; Conformance testing; Context (archaeology); Test (biology); Algorithm; Fault coverage; State (computer science); Mathematics; Engineering","score_opus":0.08039159151183822,"score_gpt":0.312587205035642,"score_spread":0.2321956135238038,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1601919490","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051667564,0.00028161416,0.937378,0.00021562928,0.00011309199,0.00013967475,0.00008669725,0.0047208816,0.005396874],"genre_scores_gemma":[0.3451596,0.0002753502,0.6467239,0.0002102078,0.0001034944,0.00016342782,0.000342423,0.0007167928,0.00630481],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.997673,0.0005739658,0.00021301392,0.00035534886,0.0009924446,0.00019218875],"domain_scores_gemma":[0.98668087,0.007788742,0.0010484091,0.0024674875,0.0017062449,0.00030830872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013782858,0.0010330392,0.00069522945,0.0014364847,0.0003824787,0.0006010589,0.0014454466,0.0007623258,0.007066689],"category_scores_gemma":[0.009956163,0.00053436705,0.0006932869,0.000939459,0.0007294877,0.0020498382,0.0014865693,0.0015032322,0.001345797],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007398464,0.00039315922,0.0015840072,0.0004306531,0.000053320317,0.00034343568,0.00022523856,0.042455114,0.098858,0.039735645,0.0033670154,0.8118146],"study_design_scores_gemma":[0.00026366557,0.0015252321,0.0015653665,0.00023885895,0.00020642146,0.0014747219,0.00015629389,0.57070655,0.26283196,0.14056526,0.020371662,0.000093979506],"about_ca_topic_score_codex":0.0007606561,"about_ca_topic_score_gemma":0.0014593711,"teacher_disagreement_score":0.007066689,"about_ca_system_score_codex":0.0005489544,"about_ca_system_score_gemma":0.0012303612,"threshold_uncertainty_score":0.023640454},"labels":[],"label_agreement":null},{"id":"W1602279527","doi":"10.1007/11902140_106","title":"Test Suite Reduction Based on Dependence Analysis","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Test suite; Computer science; Java; Test case; Suite; Equivalence (formal languages); Programming language; Test (biology); Algorithm; Mathematics; Machine learning; Discrete mathematics","score_opus":0.01573036673521473,"score_gpt":0.24977390180548278,"score_spread":0.23404353507026807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1602279527","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061051544,0.0005930676,0.9117873,0.00042520382,0.00014904636,0.0002332023,0.00033368342,0.01040047,0.015026543],"genre_scores_gemma":[0.52780944,0.00040321058,0.4562683,0.0003299842,0.0001236599,0.00031339875,0.001631689,0.0018053186,0.011315015],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99845016,0.00035892508,0.00008058919,0.00016775963,0.0007485094,0.00019399641],"domain_scores_gemma":[0.996405,0.0019853204,0.00017472479,0.00078707223,0.0005485356,0.00009932189],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006053307,0.0012819867,0.0012245761,0.0036557193,0.00048210996,0.0007421969,0.0017585761,0.0005990726,0.0060132556],"category_scores_gemma":[0.0044075553,0.0005535867,0.0017255314,0.0018485385,0.0007766115,0.0012543483,0.0013302554,0.0018719381,0.0011650831],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005037093,0.00075441273,0.0019018878,0.00036730236,0.00014128353,0.00041580023,0.00013690902,0.049869657,0.061814707,0.035323936,0.013914939,0.8348555],"study_design_scores_gemma":[0.00020854121,0.0004894437,0.0034128947,0.00009669061,0.00035259614,0.0011326545,0.0000693444,0.79849577,0.078056484,0.10789404,0.009702823,0.00008876341],"about_ca_topic_score_codex":0.001616987,"about_ca_topic_score_gemma":0.002492794,"teacher_disagreement_score":0.0060132556,"about_ca_system_score_codex":0.00067417923,"about_ca_system_score_gemma":0.0011135096,"threshold_uncertainty_score":0.02011633},"labels":[],"label_agreement":null},{"id":"W1603540703","doi":"","title":"Proceedings of the ACM SIGPLAN International Workshop on State of the Art in Java Program analysis","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; McGill University","funders":"","keywords":"Java; Computer science; State (computer science); Programming language; Software engineering","score_opus":0.02634602667210924,"score_gpt":0.295679964043182,"score_spread":0.26933393737107275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1603540703","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020419814,0.040850226,0.88820606,0.013934104,0.009126721,0.00033838517,0.00082485494,0.0055393274,0.020760497],"genre_scores_gemma":[0.18066785,0.06043175,0.65857327,0.0034224568,0.012967268,0.00051816006,0.0070882975,0.005655542,0.07067538],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98897296,0.005375927,0.001060273,0.0014052177,0.0024604243,0.0007252028],"domain_scores_gemma":[0.95943993,0.025547257,0.0007214089,0.0069523263,0.0049700704,0.002369075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018343244,0.0022848588,0.0035183476,0.002855707,0.0014810362,0.009338838,0.003916901,0.002473116,0.021685632],"category_scores_gemma":[0.02716811,0.0024014865,0.0024065159,0.002296418,0.004278505,0.00964066,0.0035326353,0.009190663,0.004745934],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017466634,0.0014595228,0.0053672306,0.00083775935,0.00035772345,0.0003223259,0.0010420383,0.007818359,0.006433381,0.039110158,0.17338766,0.7621172],"study_design_scores_gemma":[0.0006700081,0.0009984544,0.010931482,0.0019027126,0.0009871883,0.0018748,0.0010014025,0.17753619,0.025641533,0.18060319,0.59747946,0.0003735778],"about_ca_topic_score_codex":0.011934242,"about_ca_topic_score_gemma":0.01854696,"teacher_disagreement_score":0.021685632,"about_ca_system_score_codex":0.0023797783,"about_ca_system_score_gemma":0.0062664007,"threshold_uncertainty_score":0.09700948},"labels":[],"label_agreement":null},{"id":"W1607747749","doi":"","title":"Specification-based regression test selection with risk analysis","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":143,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Regression testing; Computer science; Test suite; Test Management Approach; Test case; Regression analysis; Traceability; Risk-based testing; Data mining; Unified Modeling Language; Reliability engineering; Machine learning; Software engineering; Programming language; Software; Engineering; Software development","score_opus":0.019136167122428327,"score_gpt":0.22858352951678404,"score_spread":0.2094473623943557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1607747749","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009535216,0.000038853676,0.9874514,0.00009063936,0.000009381184,0.00013971626,0.00003787582,0.0017290448,0.0009678466],"genre_scores_gemma":[0.24947007,0.000067726396,0.7478208,0.00012944813,0.000031252508,0.00046906286,0.00042261093,0.00045543548,0.0011335918],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98098433,0.009761924,0.0011021924,0.0010806911,0.006491285,0.00057967583],"domain_scores_gemma":[0.96178234,0.024435671,0.0027134705,0.004504957,0.0062499186,0.00031365856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008674227,0.0014696899,0.00092187454,0.002945475,0.00042772337,0.0014374494,0.0017810684,0.0010297456,0.0028955678],"category_scores_gemma":[0.044247918,0.0005363142,0.0015053722,0.0016346982,0.0007565067,0.0017520313,0.0014745248,0.0013627089,0.000984156],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005602433,0.0006289987,0.022404173,0.00036090365,0.00033279008,0.0009759027,0.00048540885,0.27992406,0.025086321,0.049705226,0.005588543,0.61394745],"study_design_scores_gemma":[0.00008255549,0.00024130382,0.0015036566,0.000030316383,0.00007853804,0.00041133453,0.000049562947,0.9673612,0.014641143,0.013489642,0.0020762368,0.0000344787],"about_ca_topic_score_codex":0.0016361302,"about_ca_topic_score_gemma":0.0012335777,"teacher_disagreement_score":0.008674227,"about_ca_system_score_codex":0.00077376416,"about_ca_system_score_gemma":0.0016590283,"threshold_uncertainty_score":0.04587418},"labels":[],"label_agreement":null},{"id":"W1618890699","doi":"10.1007/3-540-44830-6_12","title":"Generating Checking Sequences for a Distributed Test Architecture","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Observability; Controllability; Computer science; Sequence (biology); Communication source; Test (biology); Current (fluid); Sequence diagram; Algorithm; Programming language; Mathematics; Software; Unified Modeling Language; Telecommunications; Engineering","score_opus":0.02648434062110274,"score_gpt":0.26619954363372705,"score_spread":0.2397152030126243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1618890699","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08872952,0.000047362515,0.8956545,0.00022126218,0.00004401203,0.00031124626,0.00032793495,0.010602732,0.0040614037],"genre_scores_gemma":[0.48724425,0.00004440126,0.5067226,0.00010020837,0.000021752978,0.00026766432,0.00083153066,0.0011681664,0.0035993056],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99848986,0.00047083947,0.000100957084,0.0003046972,0.00045440852,0.00017919073],"domain_scores_gemma":[0.9903009,0.00708763,0.0005052597,0.0010526753,0.00087608397,0.00017748366],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014913568,0.0012092913,0.00062277453,0.0017101769,0.00071699184,0.0009972636,0.0014700853,0.0015721485,0.009536712],"category_scores_gemma":[0.008514616,0.0007475801,0.001077982,0.00091803307,0.0012753141,0.0014811031,0.001546865,0.0011160993,0.0014558564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017165687,0.00067516696,0.009924615,0.00072050816,0.0001387564,0.0018562233,0.000994178,0.27636746,0.07463421,0.08895928,0.0118021965,0.53221095],"study_design_scores_gemma":[0.00034729295,0.00048237504,0.0011307471,0.00009903268,0.00013847995,0.0004099963,0.00014012982,0.8376054,0.05813731,0.09614379,0.0053093606,0.0000561632],"about_ca_topic_score_codex":0.0024238105,"about_ca_topic_score_gemma":0.0044648326,"teacher_disagreement_score":0.009536712,"about_ca_system_score_codex":0.00093350007,"about_ca_system_score_gemma":0.0016774158,"threshold_uncertainty_score":0.031903505},"labels":[],"label_agreement":null},{"id":"W1623765515","doi":"10.1109/qsic.2004.8","title":"Antipattern-based detection of deficiencies in Java multithreaded software","year":2004,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Systems, Applications & Products in Data Processing (Canada); Computer Research Institute of Montréal","funders":"","keywords":"Multithreading; Java; Computer science; Programming language; Template; Software; Real time Java; Operating system; Software engineering; Computer architecture; Thread (computing)","score_opus":0.019939213965241606,"score_gpt":0.24618437800177057,"score_spread":0.22624516403652895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1623765515","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45709932,0.00033797324,0.5363736,0.00016667234,0.00003605502,0.00016270152,0.00028414073,0.004256126,0.0012833886],"genre_scores_gemma":[0.7307001,0.00015186657,0.2673968,0.00009070452,0.000019354105,0.00012984684,0.00037567393,0.0003393237,0.0007962895],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99687034,0.0006301176,0.0003632897,0.00051640224,0.0014364579,0.00018343153],"domain_scores_gemma":[0.9760418,0.011614649,0.0051496653,0.0031127783,0.0037350738,0.00034606716],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016908202,0.00050210144,0.00047391752,0.002977046,0.0004151711,0.0007567468,0.0009914307,0.00081786345,0.0009233627],"category_scores_gemma":[0.014505931,0.00029135338,0.00049780885,0.0011502563,0.0006901392,0.0018759363,0.0008032056,0.0006609339,0.0002832402],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007696618,0.00037185027,0.28992194,0.0007391551,0.00016163851,0.0017189477,0.0013992636,0.022988098,0.20225856,0.005973406,0.0014204729,0.47227702],"study_design_scores_gemma":[0.00007628601,0.0011081949,0.13477756,0.0001925103,0.0002450342,0.0073054144,0.0005435279,0.54700506,0.28806314,0.0123368325,0.008204832,0.00014162814],"about_ca_topic_score_codex":0.0010493406,"about_ca_topic_score_gemma":0.0015076927,"teacher_disagreement_score":0.002977046,"about_ca_system_score_codex":0.0003902767,"about_ca_system_score_gemma":0.000574615,"threshold_uncertainty_score":0.008942008},"labels":[],"label_agreement":null},{"id":"W163528057","doi":"10.1007/978-3-319-11743-0_11","title":"Acceptance Test Optimization","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Redundancy (engineering); Integration testing; Test (biology); Test case; System under test; Acceptance testing; Test Management Approach; Reliability engineering; Test strategy; System testing; Programming language; Software; Software engineering; Machine learning; Software system; Engineering; Operating system; Software construction","score_opus":0.01633792998939871,"score_gpt":0.24762005690595135,"score_spread":0.23128212691655264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W163528057","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014131954,0.0011072344,0.89037335,0.00074812124,0.00033496314,0.0002799826,0.00034071473,0.0063157063,0.08636796],"genre_scores_gemma":[0.5143677,0.0009864905,0.37001663,0.000855931,0.00037242722,0.0006295258,0.0020379508,0.0046520787,0.10608128],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99537754,0.0014093864,0.00015240845,0.00066273584,0.0019252973,0.0004726579],"domain_scores_gemma":[0.9945528,0.002493163,0.00022225028,0.0013806395,0.0012225065,0.00012864846],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016266899,0.0018440859,0.001490654,0.0015463311,0.00064753124,0.0020107955,0.0022848046,0.00116167,0.031288553],"category_scores_gemma":[0.0098680705,0.00065821176,0.0014291267,0.001296113,0.0009912981,0.0018268818,0.00181977,0.002979497,0.00992616],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043596982,0.00033740344,0.001108944,0.00034159364,0.00009869178,0.00021274349,0.00012617889,0.044972543,0.012412675,0.12192366,0.0402204,0.7778092],"study_design_scores_gemma":[0.0001589013,0.00047795914,0.0020227567,0.00021622426,0.00021358048,0.00090543024,0.00014214142,0.5673885,0.03901792,0.29892874,0.09042339,0.00010452243],"about_ca_topic_score_codex":0.0013129539,"about_ca_topic_score_gemma":0.0015843222,"teacher_disagreement_score":0.031288553,"about_ca_system_score_codex":0.0009204245,"about_ca_system_score_gemma":0.0014275665,"threshold_uncertainty_score":0.1046707},"labels":[],"label_agreement":null},{"id":"W1637346","doi":"10.5753/sbac-pad.2001.22210","title":"Using the SGI Pro64 Open Source Compiler Infra-Structure for Teaching and Research","year":2001,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Compiler; Computer science; Suite; Compiler construction; Compiler correctness; Interprocedural optimization; Class (philosophy); Optimizing compiler; Programming language; Source code; Parallel computing; Loop optimization; Artificial intelligence","score_opus":0.1797832101432179,"score_gpt":0.43337448754289815,"score_spread":0.25359127739968024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1637346","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10767162,0.0004413576,0.6903509,0.0018513547,0.00043815732,0.00071208726,0.0018389884,0.069053784,0.12764175],"genre_scores_gemma":[0.2228477,0.0005370677,0.6837233,0.00044251082,0.00011269861,0.000504161,0.00500744,0.020910379,0.06591472],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983188,0.00037375747,0.00008902308,0.00024826487,0.00082616333,0.00014405596],"domain_scores_gemma":[0.9920648,0.0023105878,0.00046359186,0.002677212,0.0019215614,0.00056232593],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022518525,0.00070616836,0.00037752147,0.00082032045,0.00080574414,0.0014994186,0.0010906772,0.0006199228,0.019855164],"category_scores_gemma":[0.009303264,0.0005515135,0.0004741008,0.0011940502,0.00053863414,0.002177447,0.0016107277,0.0022839091,0.015487129],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006484812,0.0009157203,0.0061491043,0.00025795022,0.000026677613,0.0003506855,0.002630604,0.006231258,0.0744614,0.02628915,0.1497454,0.7322935],"study_design_scores_gemma":[0.0002285159,0.00097279856,0.00899481,0.00011179766,0.00005048832,0.0010153691,0.0005847209,0.042313144,0.12528707,0.016261732,0.80407435,0.000105162384],"about_ca_topic_score_codex":0.00091006217,"about_ca_topic_score_gemma":0.0014051106,"teacher_disagreement_score":0.019855164,"about_ca_system_score_codex":0.00059277547,"about_ca_system_score_gemma":0.0018819897,"threshold_uncertainty_score":0.066422164},"labels":[],"label_agreement":null},{"id":"W1647188876","doi":"10.1007/978-90-481-3660-5_48","title":"A Survey of Using Model-Based Testing to Improve Quality Attributes in Distributed Systems","year":2009,"lang":"en","type":"book-chapter","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Correctness; Computer science; Reliability engineering; Reliability (semiconductor); Conformance testing; Quality (philosophy); Engineering; Programming language; Operating system; Standardization","score_opus":0.17022237723475778,"score_gpt":0.33347656235550827,"score_spread":0.1632541851207505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1647188876","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018985523,0.60191077,0.34010345,0.0027398416,0.0005035326,0.00024798373,0.000246198,0.0016839228,0.03357884],"genre_scores_gemma":[0.13559557,0.5093035,0.33881658,0.0012123574,0.00060936774,0.00020130156,0.0008858283,0.00065776135,0.012717698],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99649686,0.000807845,0.000308888,0.00039659927,0.0018617515,0.0001282202],"domain_scores_gemma":[0.9881253,0.00889775,0.00047524026,0.0007789547,0.001579866,0.0001427874],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036487628,0.00097624573,0.0011794504,0.0039410675,0.00038070767,0.0020015426,0.0026732273,0.0011205657,0.0037624044],"category_scores_gemma":[0.011640688,0.0008448742,0.0008929244,0.008459031,0.00084444607,0.004530823,0.0009241227,0.0015101474,0.0009039578],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000041350213,0.00016953764,0.0010831029,0.0016823681,0.000038586142,0.000032720287,0.00010966641,0.004384919,0.0014757618,0.009919712,0.0046785627,0.97638375],"study_design_scores_gemma":[0.0001900846,0.0012897757,0.018692609,0.011308144,0.00069476606,0.003627026,0.0007181967,0.15217806,0.03593322,0.10685416,0.6681635,0.00035037345],"about_ca_topic_score_codex":0.0022767014,"about_ca_topic_score_gemma":0.0029761812,"teacher_disagreement_score":0.0039410675,"about_ca_system_score_codex":0.0010808907,"about_ca_system_score_gemma":0.0009776107,"threshold_uncertainty_score":0.019296706},"labels":[],"label_agreement":null},{"id":"W1656270692","doi":"10.1007/978-3-642-36757-1_2","title":"Identification and Selection of Interaction Test Scenarios for Integration Testing","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Integration testing; Unit testing; Computer science; Interoperability; Test (biology); Context (archaeology); Test case; Test Management Approach; Identification (biology); Model-based testing; Reliability engineering; Data mining; Machine learning; Programming language; Engineering; Software; Software development; Software construction; Operating system","score_opus":0.030546618663785892,"score_gpt":0.2768479638866936,"score_spread":0.2463013452229077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1656270692","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30334836,0.00094299065,0.6762999,0.00041342503,0.00006947937,0.0009855295,0.0011456287,0.007055642,0.009739005],"genre_scores_gemma":[0.70156837,0.00019811235,0.29420263,0.000085685715,0.000025993475,0.00036142374,0.0023036895,0.00033937028,0.0009147082],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99516726,0.0019795687,0.00043244497,0.0006641059,0.0013549802,0.00040163216],"domain_scores_gemma":[0.9829488,0.011280734,0.0013177582,0.0015856983,0.0020825488,0.0007844619],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002492403,0.0019710644,0.0010032271,0.0046968427,0.00047414756,0.0021269915,0.0016452327,0.0015395728,0.003968753],"category_scores_gemma":[0.020763446,0.0005426839,0.0013465586,0.0011902121,0.0005843594,0.0021238695,0.0015795842,0.0009303887,0.0010948875],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033948172,0.0018728956,0.061259408,0.001034106,0.00030716375,0.0033276444,0.0008210284,0.08185403,0.11755724,0.018512301,0.008487282,0.7015721],"study_design_scores_gemma":[0.00033178562,0.0016902947,0.016437871,0.000501107,0.00044397503,0.00349446,0.00075441506,0.8562988,0.07936658,0.031304546,0.009227904,0.00014835782],"about_ca_topic_score_codex":0.0006306256,"about_ca_topic_score_gemma":0.00093526905,"teacher_disagreement_score":0.0046968427,"about_ca_system_score_codex":0.00052895787,"about_ca_system_score_gemma":0.0010060708,"threshold_uncertainty_score":0.013276756},"labels":[],"label_agreement":null},{"id":"W1672898643","doi":"10.1002/smr.1559","title":"Regression test suite selection using dependence analysis","year":2012,"lang":"en","type":"article","venue":"Journal of Software Evolution and Process","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Test suite; Regression testing; Computer science; Suite; Test case; Data mining; Representation (politics); Process (computing); Regression analysis; Machine learning; Software; Software system; Programming language","score_opus":0.024025132744568693,"score_gpt":0.30849297701315503,"score_spread":0.28446784426858635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1672898643","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1431486,0.00030003075,0.844875,0.00025871096,0.00002658514,0.0004472767,0.0005629575,0.006298055,0.0040827864],"genre_scores_gemma":[0.6845011,0.00011604931,0.3110183,0.00009493047,0.000032583062,0.00044008726,0.0021444303,0.0005649574,0.0010876594],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9951988,0.0021598965,0.00027895122,0.0005843643,0.0014679447,0.00031021918],"domain_scores_gemma":[0.97977096,0.015110731,0.0010472952,0.0014555175,0.0023409682,0.00027451894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033195193,0.0013646276,0.0010879656,0.005327622,0.000406956,0.00084776443,0.0012416791,0.00066392607,0.002658814],"category_scores_gemma":[0.019846546,0.0005154656,0.0015529467,0.0014746821,0.0005262385,0.00090606057,0.0010753475,0.0008351233,0.0006003378],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077166135,0.0004700626,0.023019738,0.00036839524,0.0002998985,0.00110588,0.00024122217,0.3681936,0.04528749,0.01105975,0.0042578974,0.5449244],"study_design_scores_gemma":[0.00006691195,0.0002644945,0.002938702,0.000045891302,0.00009664538,0.00024834997,0.000035274847,0.9676724,0.017481787,0.009325353,0.0017969793,0.000027180673],"about_ca_topic_score_codex":0.0016426475,"about_ca_topic_score_gemma":0.0016998361,"teacher_disagreement_score":0.005327622,"about_ca_system_score_codex":0.000769883,"about_ca_system_score_gemma":0.0011720334,"threshold_uncertainty_score":0.017555475},"labels":[],"label_agreement":null},{"id":"W1724543665","doi":"10.1007/978-3-642-05031-2_3","title":"Testing k-Safe Petri Nets","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Petri net; Computer science; Conformance testing; Formal specification; Programming language; Formal methods; Process architecture; Finite-state machine; Formal verification; Theoretical computer science; Operating system","score_opus":0.03099014295615103,"score_gpt":0.2624275698264454,"score_spread":0.23143742687029434,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1724543665","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13511725,0.0005476644,0.82814246,0.00044928055,0.000189177,0.00010284728,0.00020238325,0.0039138,0.031335145],"genre_scores_gemma":[0.87103945,0.00038667527,0.117135786,0.00012582105,0.00002869973,0.000059739836,0.00045006783,0.00050106423,0.010272778],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9976816,0.0004315534,0.00017101482,0.00033503614,0.0011323906,0.0002483891],"domain_scores_gemma":[0.9937337,0.004069703,0.00026341228,0.00089248444,0.0008416111,0.00019909054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001061125,0.0010970666,0.00060852815,0.0007959532,0.00055783597,0.0013095477,0.002509455,0.000980351,0.0039348747],"category_scores_gemma":[0.0067770816,0.0007062575,0.00075228827,0.00066533155,0.0016932752,0.0035108426,0.0015610891,0.0014100664,0.0011837318],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009947282,0.00032310918,0.0074224835,0.00084918406,0.0001711553,0.0026276219,0.0014729407,0.19492562,0.10454167,0.2043218,0.005364363,0.47698537],"study_design_scores_gemma":[0.00008735589,0.00027993042,0.0015179793,0.00024937643,0.00010036839,0.001166822,0.00048607594,0.40520856,0.12603904,0.45152736,0.013259535,0.00007749665],"about_ca_topic_score_codex":0.0018681127,"about_ca_topic_score_gemma":0.0022964834,"teacher_disagreement_score":0.0039348747,"about_ca_system_score_codex":0.000992224,"about_ca_system_score_gemma":0.0010505185,"threshold_uncertainty_score":0.013163447},"labels":[],"label_agreement":null},{"id":"W172667635","doi":"","title":"An Algebraic Query Method for Static Program Analysis and Measurement.","year":2008,"lang":"en","type":"article","venue":"Software Engineering and Data Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Query optimization; Query language; Algebraic number; Programming language; Theoretical computer science; Information retrieval; Mathematics","score_opus":0.04551195105983421,"score_gpt":0.3039288780590528,"score_spread":0.2584169269992186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W172667635","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00069824065,0.00010258693,0.991905,0.00009363148,0.000037154347,0.00007090913,0.00018978836,0.0055765924,0.0013260962],"genre_scores_gemma":[0.05656659,0.0003151655,0.9350242,0.0002196057,0.0001723797,0.00046736756,0.0010881851,0.0018480569,0.004298406],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9892083,0.0022469228,0.0011882101,0.0016527189,0.005218364,0.0004854985],"domain_scores_gemma":[0.98967314,0.0040961993,0.0006044764,0.0030167017,0.0023323172,0.0002771459],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051495074,0.001377604,0.001783252,0.0036265757,0.0013360153,0.004310476,0.0039261673,0.0013466239,0.014167942],"category_scores_gemma":[0.02683864,0.0012171024,0.0019818246,0.0047782464,0.0023361682,0.00838291,0.0054696817,0.002888051,0.0053999065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000418758,0.00015630461,0.0016677581,0.00074983865,0.0001739374,0.0002124715,0.0007203282,0.0059572486,0.011723707,0.44208708,0.030182706,0.50594985],"study_design_scores_gemma":[0.0002712445,0.00028684345,0.0013869229,0.0002402305,0.0003227004,0.0011716578,0.00045309498,0.26958093,0.030706704,0.54853034,0.14685813,0.00019116448],"about_ca_topic_score_codex":0.0039142063,"about_ca_topic_score_gemma":0.0039354707,"teacher_disagreement_score":0.014167942,"about_ca_system_score_codex":0.0012526822,"about_ca_system_score_gemma":0.0030556994,"threshold_uncertainty_score":0.04739648},"labels":[],"label_agreement":null},{"id":"W1748160142","doi":"","title":"Clara: partially evaluating runtime monitors at compile time tutorial supplement","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"AspectJ; Computer science; Compile time; Compiler; Runtime verification; Static analysis; Runtime system; Context (archaeology); Programming language; Property (philosophy); Residual; State (computer science); Aspect-oriented programming; Software; Formal verification; Algorithm","score_opus":0.021986125714112147,"score_gpt":0.3046886914802215,"score_spread":0.28270256576610936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1748160142","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0054468284,0.0020549851,0.9050452,0.0020554606,0.0022408303,0.00024199043,0.0019364137,0.044183694,0.036794532],"genre_scores_gemma":[0.097308986,0.002026634,0.79571915,0.0013792556,0.0021600088,0.0005045179,0.005152241,0.01814872,0.07760049],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9976991,0.0005421765,0.00015061427,0.00046640908,0.0009847593,0.00015694194],"domain_scores_gemma":[0.9936724,0.0031623486,0.00029231425,0.0007589639,0.0018816517,0.0002323338],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022446646,0.0018344248,0.0014253644,0.0019377384,0.00085650117,0.002936494,0.0019152393,0.0012316885,0.06208399],"category_scores_gemma":[0.011357382,0.0012894204,0.0013320467,0.0008382294,0.00085367827,0.0036782625,0.001656642,0.0025996517,0.017976366],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049978506,0.00020173739,0.0008674095,0.0007830587,0.00009023223,0.00056150375,0.0004867924,0.012220708,0.025156334,0.11559324,0.3713564,0.47218275],"study_design_scores_gemma":[0.000135355,0.00022906525,0.0016110323,0.00052506867,0.00012267003,0.0011465756,0.00016228341,0.12794377,0.047961175,0.09049596,0.72944164,0.00022544467],"about_ca_topic_score_codex":0.0026539252,"about_ca_topic_score_gemma":0.003878792,"teacher_disagreement_score":0.06208399,"about_ca_system_score_codex":0.0017726228,"about_ca_system_score_gemma":0.0017280508,"threshold_uncertainty_score":0.20769161},"labels":[],"label_agreement":null},{"id":"W1757839379","doi":"10.1007/978-3-642-23716-4_19","title":"Divide-by-Zero Exception Raising via Branch Coverage","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Testability; Hill climbing; Simulated annealing; Computer science; Genetic programming; Constraint programming; Algorithm; Mathematical optimization; Mathematics; Artificial intelligence","score_opus":0.02637198270498408,"score_gpt":0.2512384320766876,"score_spread":0.22486644937170353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1757839379","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.081992574,0.00079853454,0.87816167,0.00053594256,0.00023167784,0.00007445017,0.000108876906,0.007275037,0.030821279],"genre_scores_gemma":[0.8387461,0.0003652802,0.15033878,0.00030780895,0.000111597285,0.00006814223,0.0002037182,0.0009191445,0.008939502],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998403,0.0003247895,0.0001007991,0.0002863372,0.0005829647,0.0003021478],"domain_scores_gemma":[0.99589455,0.0019979328,0.0001857932,0.0014372863,0.000338457,0.00014591323],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010959457,0.0008951541,0.000997139,0.0015832674,0.00065997115,0.0015163298,0.001578879,0.0010367937,0.0059996415],"category_scores_gemma":[0.0057897717,0.00065889774,0.00074212725,0.0012935073,0.0013248306,0.0034491906,0.0032723632,0.0023260908,0.0014390197],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005654516,0.00019981273,0.0025475675,0.00033672398,0.000068625064,0.00077117264,0.00046669896,0.022585742,0.026301693,0.18772711,0.009146406,0.7492829],"study_design_scores_gemma":[0.00010518291,0.0002498314,0.0012481229,0.0002058639,0.00026812791,0.0013624674,0.00014751319,0.23928092,0.067460634,0.667592,0.02198138,0.00009793406],"about_ca_topic_score_codex":0.0003675384,"about_ca_topic_score_gemma":0.00077558716,"teacher_disagreement_score":0.0059996415,"about_ca_system_score_codex":0.0004615601,"about_ca_system_score_gemma":0.0006946659,"threshold_uncertainty_score":0.02007085},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W176867143","doi":"10.1007/978-3-642-33119-0_16","title":"Boosting Search Based Testing by Using Constraint Based Testing","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Population; Boosting (machine learning); Selection (genetic algorithm); Constraint (computer-aided design); Test case; Evolutionary computation; Evolutionary algorithm; Mathematical optimization; Machine learning; Algorithm; Mathematics","score_opus":0.07689082099450231,"score_gpt":0.29115087053938743,"score_spread":0.21426004954488512,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W176867143","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03188239,0.00044218323,0.95847243,0.00020080662,0.0000935416,0.00011647104,0.00010262822,0.002593767,0.0060958434],"genre_scores_gemma":[0.5416528,0.0001553286,0.45270973,0.00029771603,0.000104744475,0.00018706836,0.00042033664,0.0006460648,0.0038261448],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968425,0.00089107157,0.00013806537,0.00041822475,0.0013934099,0.00031672997],"domain_scores_gemma":[0.99045765,0.006127492,0.00038434286,0.0012946163,0.0014738456,0.00026202857],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023101126,0.001147723,0.0018265609,0.0017172121,0.00044362003,0.0009547781,0.0034387738,0.0015945217,0.006343013],"category_scores_gemma":[0.011800775,0.0006067962,0.0010966674,0.001931028,0.0009985213,0.0020512042,0.0017690656,0.0015838855,0.0010566784],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005595218,0.00040816786,0.0038157708,0.00027943891,0.0001364375,0.00031218556,0.0000731776,0.42340484,0.01923904,0.03996909,0.0073813116,0.50442094],"study_design_scores_gemma":[0.000021226266,0.000055553533,0.00027452188,0.000009838847,0.000018817684,0.00006701572,0.0000036734912,0.9861691,0.0019994685,0.010619311,0.0007542145,0.0000071943628],"about_ca_topic_score_codex":0.002688286,"about_ca_topic_score_gemma":0.0030370366,"teacher_disagreement_score":0.006343013,"about_ca_system_score_codex":0.00083522557,"about_ca_system_score_gemma":0.0012306818,"threshold_uncertainty_score":0.021219492},"labels":[],"label_agreement":null},{"id":"W1769260096","doi":"10.1007/3-540-45745-3_15","title":"Towards Specification-Based Web Testing","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Web testing; Correctness; Robustness (evolution); Software engineering; Non-regression testing; Web application; Software performance testing; Web modeling; Robustness testing; White-box testing; Reliability engineering; Web service; Programming language; Web application security; Web development; World Wide Web; Software construction; Software development; Software","score_opus":0.04896644395236944,"score_gpt":0.2608482569094928,"score_spread":0.21188181295712338,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1769260096","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021853514,0.0001616004,0.99171185,0.0002491461,0.000036241534,0.000054922442,0.000021759495,0.0015520179,0.004027135],"genre_scores_gemma":[0.095374964,0.0005782383,0.89485675,0.00034456622,0.000050629023,0.00022712283,0.0003962521,0.00085983804,0.0073116072],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9916721,0.0030555683,0.0005252572,0.0006466593,0.0037071572,0.00039321926],"domain_scores_gemma":[0.98382866,0.008604982,0.0004609245,0.004195971,0.002634562,0.0002749145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061841654,0.0012411937,0.0009827394,0.0019796847,0.00066840317,0.0038198193,0.003907801,0.002341529,0.0064418125],"category_scores_gemma":[0.018147595,0.0017679604,0.0013674559,0.0013745249,0.0028286388,0.0063053053,0.0040686014,0.0054405676,0.0034504724],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019308753,0.00043129473,0.0015473949,0.0005219283,0.00009313625,0.0005022836,0.0009207049,0.050824657,0.017616222,0.465674,0.0117704915,0.4499047],"study_design_scores_gemma":[0.000073694406,0.00010210965,0.0003247722,0.00026613584,0.000073611656,0.00044291047,0.00019975156,0.5553068,0.02676932,0.38159478,0.034803614,0.000042529726],"about_ca_topic_score_codex":0.001971145,"about_ca_topic_score_gemma":0.0022630084,"teacher_disagreement_score":0.0064418125,"about_ca_system_score_codex":0.0010738682,"about_ca_system_score_gemma":0.0015356314,"threshold_uncertainty_score":0.032705367},"labels":[],"label_agreement":null},{"id":"W1773476055","doi":"10.5281/zenodo.1079180","title":"Generating State-Based Testing Models For Object-Oriented Framework Interface Classes","year":2008,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Interface (matter); Class (philosophy); Implementation; Object-oriented programming; State (computer science); Software; Object (grammar); Programming language; Model-based testing; White-box testing; Test case; Software engineering; Software system; Software construction; Artificial intelligence; Operating system; Machine learning","score_opus":0.07828415661528232,"score_gpt":0.2790379239856586,"score_spread":0.20075376737037626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1773476055","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036173258,0.00004327337,0.9574076,0.00009262711,0.000017101058,0.0002421519,0.00030048558,0.0037598528,0.0019637074],"genre_scores_gemma":[0.3598658,0.0001440349,0.6344071,0.00006863619,0.000013709745,0.00076279894,0.0017295697,0.0005786853,0.002429618],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987463,0.0003599795,0.00010495496,0.0001808787,0.00047975965,0.00012798031],"domain_scores_gemma":[0.995839,0.0027558617,0.00037350136,0.00052202115,0.0004385537,0.00007110284],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014896635,0.0009509144,0.00052776106,0.0015069403,0.0006094723,0.0013119471,0.0015767082,0.0014217177,0.0025479782],"category_scores_gemma":[0.006768472,0.00072201603,0.001974664,0.00059834076,0.0010240254,0.0016790448,0.001085541,0.0009775525,0.00051409786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028089987,0.00032975347,0.0051014996,0.00025564226,0.000069429,0.00074753986,0.0009538571,0.7624122,0.015949566,0.12507337,0.0018398161,0.08698645],"study_design_scores_gemma":[0.00003760598,0.000059142603,0.0002621404,0.000028855486,0.000030464302,0.00007410606,0.000034093227,0.9681898,0.0115316,0.016930958,0.0028009603,0.000020331974],"about_ca_topic_score_codex":0.007160405,"about_ca_topic_score_gemma":0.0068099923,"teacher_disagreement_score":0.007160405,"about_ca_system_score_codex":0.001531108,"about_ca_system_score_gemma":0.0016343661,"threshold_uncertainty_score":0.0142374635},"labels":[],"label_agreement":null},{"id":"W1774577175","doi":"10.1007/978-3-540-45070-2_17","title":"Java Subtype Tests in Real-Time","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Defense Advanced Research Projects Agency; McGill University","keywords":"Computer science; Java; Speedup; Byte; Constant (computer programming); C dynamic memory allocation; Class (philosophy); Real time Java; Operating system; Programming language; Parallel computing; Memory management; Artificial intelligence","score_opus":0.021264871785459775,"score_gpt":0.26276019879943985,"score_spread":0.24149532701398008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1774577175","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020734191,0.00057559414,0.9269893,0.00042501834,0.00031891474,0.00010637961,0.00026526043,0.029735824,0.020849591],"genre_scores_gemma":[0.38411355,0.00042821883,0.5718042,0.0005597167,0.00017776119,0.00020949672,0.0010637164,0.008727004,0.032916285],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99331516,0.0019971947,0.00044717788,0.0008881192,0.002738769,0.00061362534],"domain_scores_gemma":[0.9807037,0.011024207,0.001155873,0.004638099,0.0020702644,0.0004078762],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004107974,0.0011517649,0.0012675195,0.0012175947,0.000583764,0.0024036225,0.005376315,0.0016723282,0.015367507],"category_scores_gemma":[0.017820757,0.0010949951,0.0008710955,0.0011732973,0.001865711,0.007130092,0.0019072318,0.0025403595,0.004820519],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001258128,0.00045091292,0.004570778,0.0008152008,0.00009073366,0.00086701236,0.0010624408,0.0244166,0.022450373,0.15641588,0.04336382,0.7442382],"study_design_scores_gemma":[0.0004940392,0.00059579354,0.0029515587,0.0005276973,0.0001940779,0.002374046,0.00058133074,0.45156696,0.11653212,0.32242417,0.101544,0.00021419111],"about_ca_topic_score_codex":0.0021291007,"about_ca_topic_score_gemma":0.0028253132,"teacher_disagreement_score":0.015367507,"about_ca_system_score_codex":0.0009274477,"about_ca_system_score_gemma":0.0011417668,"threshold_uncertainty_score":0.051409423},"labels":[],"label_agreement":null},{"id":"W1775351421","doi":"10.1007/978-3-642-20677-1_16","title":"Test-Driven Development of Graphical User Interfaces: A Pilot Evaluation","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in business information processing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Graphical user interface; Graphical user interface testing; Computer science; Human–computer interaction; User interface; Pilot test; Process (computing); Test (biology); Interface (matter); Software engineering; User interface design; User experience design; Operating system","score_opus":0.04664004837837124,"score_gpt":0.2719769454355778,"score_spread":0.22533689705720658,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1775351421","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9523544,0.00017898975,0.041746818,0.00007408233,0.000026670017,0.0017660854,0.00037177117,0.0019625556,0.0015185702],"genre_scores_gemma":[0.91165894,0.0002739115,0.08100184,0.00016245412,0.000017906521,0.0018275242,0.001427601,0.00057221623,0.0030577134],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9958776,0.0025910528,0.00024472066,0.0002682343,0.0007895828,0.0002288194],"domain_scores_gemma":[0.9687871,0.02260969,0.0006872424,0.0023975796,0.004131252,0.0013871151],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007117199,0.0012979513,0.0005703676,0.0009513837,0.00036593035,0.0006358287,0.002789575,0.0011766528,0.0032945275],"category_scores_gemma":[0.026955506,0.0006663234,0.00065577944,0.0004956987,0.0006541135,0.0010496615,0.0012865553,0.0007632377,0.000844576],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.023156846,0.055928692,0.021121493,0.0028563843,0.0005611829,0.0018132593,0.01239679,0.037428916,0.19386472,0.0013371132,0.0058691427,0.64366555],"study_design_scores_gemma":[0.024697404,0.21892941,0.13669968,0.0006295469,0.0016299207,0.0033263161,0.0042912886,0.27630824,0.3071021,0.0023457957,0.023495223,0.000545049],"about_ca_topic_score_codex":0.0030634343,"about_ca_topic_score_gemma":0.0023957083,"teacher_disagreement_score":0.007117199,"about_ca_system_score_codex":0.00058458134,"about_ca_system_score_gemma":0.0013285828,"threshold_uncertainty_score":0.037639797},"labels":[],"label_agreement":null},{"id":"W1785201090","doi":"10.1007/978-3-642-15585-7_10","title":"Linguistic Security Testing for Text Communication Protocols","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Protocol (science); Syntax; Programming language; Grammar; Formal grammar; Cryptographic protocol; Communications protocol; Language Of Temporal Ordering Specification; Formal specification; Natural language processing; Rule-based machine translation; Computer security; Computer network; Linguistics","score_opus":0.04246884196833257,"score_gpt":0.3168149333576922,"score_spread":0.2743460913893596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1785201090","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056012742,0.0007656735,0.88131857,0.0023859337,0.00029603962,0.00019938321,0.00022884308,0.0058948216,0.052897982],"genre_scores_gemma":[0.73044693,0.0005297858,0.24595112,0.0004967071,0.0002053095,0.00025083704,0.00062872353,0.0013212468,0.020169308],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99508214,0.0018089587,0.00032323535,0.0004072508,0.0019812218,0.00039712884],"domain_scores_gemma":[0.9849937,0.010520865,0.0004903122,0.0021063252,0.0016854982,0.000203303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002525857,0.00091750844,0.00074378826,0.0014746808,0.0014320688,0.0022721943,0.0018911508,0.0015531746,0.009663762],"category_scores_gemma":[0.016818369,0.0005396617,0.0009404128,0.00095459475,0.0035473392,0.0063335234,0.0023055354,0.002878946,0.0022451067],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055666827,0.00025965512,0.0012754785,0.0006023087,0.000044068787,0.00074202794,0.002075238,0.01225226,0.02594725,0.533498,0.016860263,0.40588686],"study_design_scores_gemma":[0.00007828272,0.00016929138,0.0004881489,0.0003143927,0.000071957555,0.0007788214,0.00041806602,0.1424255,0.05548181,0.7710605,0.028635606,0.00007754836],"about_ca_topic_score_codex":0.0010971779,"about_ca_topic_score_gemma":0.00081861863,"teacher_disagreement_score":0.009663762,"about_ca_system_score_codex":0.0012196457,"about_ca_system_score_gemma":0.0012111196,"threshold_uncertainty_score":0.032328486},"labels":[],"label_agreement":null},{"id":"W179471450","doi":"","title":"A new optimal algorithm for outerplanar graph testing.","year":2004,"lang":"en","type":"article","venue":"Scholarship at UWindsor (University of Windsor)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Algorithm; Outerplanar graph; Graph; Computer science; Mathematics; Combinatorics; Pathwidth; Line graph","score_opus":0.029008118115395207,"score_gpt":0.23751809974421426,"score_spread":0.20850998162881906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W179471450","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01436912,0.00064800587,0.96766424,0.0006092051,0.0001952017,0.00034103508,0.00034972435,0.0037205727,0.012102834],"genre_scores_gemma":[0.095059276,0.00031866718,0.89271283,0.00031349718,0.00007599894,0.00030591435,0.0013452321,0.0005340001,0.009334557],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989542,0.00011541394,0.000061619765,0.0003677411,0.00032581657,0.0001752547],"domain_scores_gemma":[0.99918264,0.0003017284,0.000096002324,0.0001952796,0.00016532563,0.000059039645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000479619,0.0010110051,0.0008703442,0.001243837,0.0007329786,0.001165022,0.0016393111,0.0011252799,0.012005224],"category_scores_gemma":[0.002787815,0.0005076641,0.00095680426,0.0012014809,0.0008497325,0.0031880839,0.0023062418,0.0013536299,0.0031653077],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003285132,0.00022870874,0.0012705199,0.00045030165,0.000065178196,0.0002124686,0.00023717416,0.030297318,0.020769894,0.056411184,0.030989625,0.85873914],"study_design_scores_gemma":[0.00037404138,0.00041400827,0.0015065274,0.00011520977,0.00011964153,0.001563741,0.00030179275,0.68358564,0.026137348,0.18512182,0.100665756,0.000094565665],"about_ca_topic_score_codex":0.0024011368,"about_ca_topic_score_gemma":0.0030719782,"teacher_disagreement_score":0.012005224,"about_ca_system_score_codex":0.001052193,"about_ca_system_score_gemma":0.001648173,"threshold_uncertainty_score":0.04016143},"labels":[],"label_agreement":null},{"id":"W1795706927","doi":"10.1007/978-3-540-31832-3_9","title":"An Evaluation of Auto-Scoping in OpenMP","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Ministerstvo Školství, Mládeže a Tělovýchovy","keywords":"Computer science; Compiler; Suite; Programming language; Benchmark (surveying); Speedup; Parallel computing; Fortran; Compiler construction; Reduction (mathematics)","score_opus":0.051708513546617735,"score_gpt":0.33146923269746137,"score_spread":0.27976071915084366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1795706927","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88437086,0.0022419775,0.07814755,0.00043951563,0.00021866936,0.0004993667,0.0015849635,0.009743026,0.022754047],"genre_scores_gemma":[0.93184805,0.000432164,0.060259137,0.0000856602,0.00004405311,0.00018300704,0.0018690211,0.0013603372,0.003918483],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9850723,0.0073933695,0.0012812677,0.0008649456,0.004665411,0.00072270085],"domain_scores_gemma":[0.78504825,0.17991516,0.0035767923,0.018230174,0.012302396,0.00092721503],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009549373,0.0008288031,0.0008665361,0.0025864902,0.0011805986,0.0017819767,0.0026113794,0.0016651566,0.0059962524],"category_scores_gemma":[0.07713503,0.0006006222,0.00077597867,0.002367233,0.0013605452,0.0047571887,0.0028497633,0.0009994821,0.0007193986],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.011974747,0.0027839716,0.02744618,0.0063381027,0.00058597245,0.0007763569,0.0045311525,0.10178558,0.024221238,0.024137309,0.011445436,0.783974],"study_design_scores_gemma":[0.00196293,0.00894877,0.020395996,0.0018678955,0.0018594043,0.0013003161,0.004095895,0.75938296,0.13331495,0.035038535,0.031563744,0.00026862702],"about_ca_topic_score_codex":0.0036151423,"about_ca_topic_score_gemma":0.0042417343,"teacher_disagreement_score":0.009549373,"about_ca_system_score_codex":0.000870139,"about_ca_system_score_gemma":0.0017130018,"threshold_uncertainty_score":0.05050254},"labels":[],"label_agreement":null},{"id":"W1819615023","doi":"10.1109/aswec.2000.844580","title":"Tools and techniques for Java API testing","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Java; Software portability; Reuse; Component (thermodynamics); Software engineering; Unit testing; Programming language; JavaBeans; Object-oriented programming; Real time Java; Component-based software engineering; Software; Software development; Engineering","score_opus":0.11025616659943181,"score_gpt":0.27524183244310696,"score_spread":0.16498566584367513,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1819615023","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005484491,0.0004940751,0.99141556,0.00016774013,0.00004927719,0.00009462587,0.00005554652,0.004490286,0.0026844435],"genre_scores_gemma":[0.01698148,0.0012912959,0.97697824,0.00022613916,0.00010320313,0.00043874004,0.00041322096,0.0012373703,0.002330263],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9922637,0.0015952663,0.00071914913,0.00058128603,0.004517708,0.00032285816],"domain_scores_gemma":[0.98982024,0.0055159763,0.00074123597,0.0021171884,0.0016109716,0.0001943174],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035616695,0.0023348916,0.0015152076,0.0062547387,0.0010385326,0.002726949,0.0035227675,0.0020861698,0.006048368],"category_scores_gemma":[0.021949911,0.0014053448,0.0018691941,0.003938524,0.00214622,0.0052467096,0.0031237807,0.0044956305,0.00497333],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007238861,0.00023480051,0.0008695191,0.0008626949,0.00007878427,0.00081483414,0.00064779265,0.013065364,0.016710812,0.18022244,0.017049076,0.7693716],"study_design_scores_gemma":[0.00017678326,0.00028167065,0.0014572635,0.001195234,0.0001397528,0.0055637877,0.00022982877,0.1407503,0.03760017,0.53763926,0.27470225,0.00026368097],"about_ca_topic_score_codex":0.00085675006,"about_ca_topic_score_gemma":0.0006494057,"teacher_disagreement_score":0.0062547387,"about_ca_system_score_codex":0.0005761563,"about_ca_system_score_gemma":0.0013134811,"threshold_uncertainty_score":0.02023387},"labels":[],"label_agreement":null},{"id":"W1827797309","doi":"10.1109/hcc.2001.995275","title":"A testing methodology for a dataflow based visual programming language","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Dataflow; Computer science; Dataflow architecture; Programming language; Testability; Control flow; Reliability engineering","score_opus":0.19604624704805348,"score_gpt":0.37455743884055015,"score_spread":0.17851119179249667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1827797309","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007386397,0.0000256181,0.99105364,0.000072706855,0.000007558748,0.000118181335,0.000020333573,0.00084778754,0.0004677748],"genre_scores_gemma":[0.16390315,0.00005730864,0.8345376,0.00010464131,0.000011829509,0.000273828,0.00011017532,0.00016401836,0.0008374647],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9960938,0.0012243396,0.00041086142,0.0005940478,0.0015231943,0.00015369675],"domain_scores_gemma":[0.9893273,0.0060724407,0.000975589,0.0010770569,0.0023314643,0.00021614805],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041101985,0.00072441006,0.0004555764,0.0015654647,0.0005645687,0.00151173,0.0016844438,0.0010732394,0.0013936543],"category_scores_gemma":[0.019262668,0.0003608151,0.00092478597,0.0006014921,0.0021992947,0.0019068389,0.0010729582,0.0009424957,0.0002526758],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037689868,0.00045783323,0.008135048,0.0010539683,0.00012446081,0.0014746874,0.0017484778,0.081810065,0.1143389,0.2465491,0.0025443027,0.54138625],"study_design_scores_gemma":[0.00012798437,0.00085483416,0.0012884735,0.00028258693,0.000088208515,0.0017397667,0.00020234057,0.7763698,0.11968016,0.08585319,0.013422591,0.000090039845],"about_ca_topic_score_codex":0.0012019675,"about_ca_topic_score_gemma":0.0008612473,"teacher_disagreement_score":0.0041101985,"about_ca_system_score_codex":0.0008853201,"about_ca_system_score_gemma":0.0014523027,"threshold_uncertainty_score":0.021737099},"labels":[],"label_agreement":null},{"id":"W1833943521","doi":"10.1007/978-3-540-25940-4_65","title":"Toward an Undergraduate League for RoboCup","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"League; Computer science; Artificial intelligence; Political science; Mathematics education; Engineering; Psychology","score_opus":0.04226289565341727,"score_gpt":0.2829359307603361,"score_spread":0.2406730351069188,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1833943521","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07535826,0.0072351177,0.036301095,0.072467744,0.014994659,0.00041014643,0.0005979302,0.0021349452,0.79050004],"genre_scores_gemma":[0.10567372,0.001911546,0.01795452,0.002731972,0.0008230643,0.00008396536,0.0005729078,0.00043023235,0.86981815],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99936324,0.000101388265,0.000018352432,0.000086969034,0.00021613947,0.00021399677],"domain_scores_gemma":[0.9984043,0.00002871118,0.000020780555,0.000036452795,0.00034510755,0.0011645281],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015655115,0.00085755985,0.00032405934,0.0009305825,0.004392374,0.0049250103,0.0010995332,0.0014255694,0.04834197],"category_scores_gemma":[0.0013280318,0.00034644097,0.00031734508,0.00038044195,0.00090270897,0.002544886,0.0034898743,0.0030816067,0.018117594],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013121858,0.0005872626,0.0019643118,0.00007161006,0.0000092959435,0.00011895698,0.0025542816,0.0005881583,0.0017608557,0.08618497,0.6725835,0.23344563],"study_design_scores_gemma":[0.000008464436,0.0001460343,0.0008387644,0.00006793013,0.0000035188543,0.00007068239,0.0014027084,0.0006511657,0.0006952,0.010958681,0.9851482,0.000008660696],"about_ca_topic_score_codex":0.0068574394,"about_ca_topic_score_gemma":0.03200142,"teacher_disagreement_score":0.04834197,"about_ca_system_score_codex":0.0022329893,"about_ca_system_score_gemma":0.0036939604,"threshold_uncertainty_score":0.16171998},"labels":[],"label_agreement":null},{"id":"W1845490991","doi":"10.1007/11901433_25","title":"Conditions for Avoiding Controllability Problems in Distributed Testing","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Controllability; Computer science; Finite-state machine; Context (archaeology); Sequence (biology); Distributed computing; Conformance testing; Test (biology); State (computer science); Architecture; Test case; System under test; Algorithm; Machine learning; Operating system; Mathematics","score_opus":0.032157442413021396,"score_gpt":0.2695266468818414,"score_spread":0.23736920446882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1845490991","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03423592,0.00039698408,0.9527533,0.00066853757,0.00022121878,0.00028633382,0.000239916,0.002132859,0.009065005],"genre_scores_gemma":[0.71393263,0.00049622095,0.27639693,0.0006033836,0.0007308427,0.0010835455,0.00087335764,0.000743434,0.0051396056],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9924509,0.001865923,0.000743246,0.0013725717,0.0021850257,0.0013823307],"domain_scores_gemma":[0.8714038,0.105557956,0.0047843447,0.007601301,0.006968404,0.0036841156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008285954,0.0021954034,0.0025312833,0.0028265319,0.0017012957,0.002765095,0.0037426618,0.0035880902,0.007219259],"category_scores_gemma":[0.059219588,0.0015018035,0.002713478,0.0018532648,0.0057815784,0.010087391,0.0058758827,0.00766827,0.0011597129],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003413986,0.00080824486,0.0037382918,0.0013341327,0.00022314982,0.0017306784,0.0010234199,0.109415784,0.02958487,0.719881,0.010142716,0.118703745],"study_design_scores_gemma":[0.0006844092,0.00056842196,0.0008589108,0.00017882083,0.00019499745,0.0006086565,0.00016341121,0.20289446,0.017651806,0.7724737,0.0035921882,0.00013008487],"about_ca_topic_score_codex":0.0009982664,"about_ca_topic_score_gemma":0.00096905045,"teacher_disagreement_score":0.008285954,"about_ca_system_score_codex":0.0010256466,"about_ca_system_score_gemma":0.0019710718,"threshold_uncertainty_score":0.043820858},"labels":[],"label_agreement":null},{"id":"W1846902312","doi":"10.1007/11532378_14","title":"The Use of Traces for Inlining in Java Programs","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Java; TRACE (psycholinguistics); Programming language; Operating system; Overhead (engineering); Just-in-time compilation; Code (set theory); Virtual machine; Parallel computing","score_opus":0.0715089159885698,"score_gpt":0.29134742622707543,"score_spread":0.21983851023850565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1846902312","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040945612,0.000982704,0.940825,0.00043720196,0.00013448422,0.000058784586,0.00013243106,0.008338129,0.008145663],"genre_scores_gemma":[0.7013487,0.0010782714,0.28521886,0.00020545915,0.00008717983,0.000098554425,0.0002482869,0.0036077984,0.008106932],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99796593,0.0008884666,0.00015785873,0.00027802153,0.00057217263,0.00013755112],"domain_scores_gemma":[0.97189474,0.020066204,0.0011054494,0.005416411,0.0011861363,0.00033107816],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002233252,0.0009164878,0.00060656114,0.0017595266,0.0008635091,0.0021279673,0.0018422445,0.0011906208,0.0043655625],"category_scores_gemma":[0.022986503,0.0010360976,0.00064165116,0.0014454294,0.002006226,0.008732261,0.0017257228,0.0022398108,0.000757998],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006146304,0.00021660766,0.0042225067,0.0005841428,0.00006387084,0.0005035289,0.0024963925,0.030767944,0.0136496015,0.21569149,0.0058637457,0.7253256],"study_design_scores_gemma":[0.00007376937,0.000470171,0.001290389,0.00056073326,0.00018244467,0.0012589636,0.00053210097,0.37777883,0.09422287,0.48430246,0.039200522,0.00012665142],"about_ca_topic_score_codex":0.0017583391,"about_ca_topic_score_gemma":0.0024336444,"teacher_disagreement_score":0.0043655625,"about_ca_system_score_codex":0.0006885953,"about_ca_system_score_gemma":0.0007383976,"threshold_uncertainty_score":0.01460427},"labels":[],"label_agreement":null},{"id":"W1852182718","doi":"10.1007/978-0-387-35497-2_21","title":"Formulation of the Interaction Test Coverage Problem as an Integer Program","year":2002,"lang":"en","type":"book-chapter","venue":"IFIP advances in information and communication technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Integer programming; Completeness (order theory); Cover (algebra); Computer science; Set (abstract data type); Integer (computer science); Set cover problem; Mathematical optimization; Test (biology); Function (biology); Test case; Theoretical computer science; Mathematics; Algorithm; Engineering; Programming language; Machine learning","score_opus":0.013116465940273079,"score_gpt":0.2831618641821669,"score_spread":0.2700453982418938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1852182718","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013366346,0.00040090017,0.960831,0.0011062914,0.0000901123,0.00017577683,0.0005779928,0.00031454742,0.023137072],"genre_scores_gemma":[0.33797848,0.0010255222,0.6385446,0.00046359416,0.00032784755,0.0010067394,0.001469652,0.00035426026,0.018829273],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99843615,0.000594606,0.00006666403,0.00026723326,0.00041705935,0.00021834213],"domain_scores_gemma":[0.9958691,0.0034301968,0.00021768252,0.00011861544,0.00028279657,0.00008150128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001971719,0.0013902043,0.0011424675,0.0011835628,0.00039656894,0.0022935192,0.0016305409,0.0016739634,0.01076663],"category_scores_gemma":[0.007126727,0.0007417465,0.0011129561,0.0014137169,0.0013955542,0.0022346876,0.0012558792,0.0021330735,0.00068411074],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010255023,0.00014792985,0.00059766986,0.0003619952,0.000051492076,0.00023361649,0.00013613416,0.7687445,0.0012926265,0.16638708,0.007886538,0.054057807],"study_design_scores_gemma":[0.00004838565,0.00005349509,0.00018754356,0.000050473773,0.000028769737,0.0000845733,0.000044583918,0.9159167,0.00059116166,0.07964471,0.003334985,0.000014581376],"about_ca_topic_score_codex":0.005672589,"about_ca_topic_score_gemma":0.0048998576,"teacher_disagreement_score":0.01076663,"about_ca_system_score_codex":0.0013927944,"about_ca_system_score_gemma":0.0019092225,"threshold_uncertainty_score":0.036017954},"labels":[],"label_agreement":null},{"id":"W1867551913","doi":"10.1109/icst.2015.7102602","title":"Prioritizing Manual Test Cases in Traditional and Rapid Release Environments","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Mitacs; Aalto-Yliopisto; University of Manitoba","keywords":"Computer science; Prioritization; Test suite; Agile software development; Unit testing; Code coverage; Ranking (information retrieval); Test case; Test (biology); Reliability engineering; Test Management Approach; White-box testing; Regression testing; Test script; Software engineering; Software; Software system; Machine learning; Operating system; Engineering","score_opus":0.07290767552958227,"score_gpt":0.26705351917657033,"score_spread":0.19414584364698806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1867551913","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.59448344,0.0018467908,0.3882081,0.00061911764,0.00011617316,0.0013201943,0.00023713366,0.005476444,0.0076925885],"genre_scores_gemma":[0.7795122,0.0003937326,0.2159213,0.00024581302,0.0000712478,0.00048873865,0.00071558455,0.0007806834,0.0018706569],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.979629,0.007710273,0.0011766184,0.002133397,0.008039144,0.0013115827],"domain_scores_gemma":[0.91056055,0.06495935,0.008925024,0.0068512294,0.0068718023,0.0018321801],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011444819,0.0016237237,0.0008515432,0.0038920636,0.0004802052,0.0024420726,0.0020131136,0.00088956405,0.0015867958],"category_scores_gemma":[0.05556221,0.00086487975,0.0008063045,0.0014304651,0.0009439879,0.0025368482,0.0019536938,0.0013468439,0.00059000787],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017527863,0.0018972339,0.042992547,0.00090683374,0.00026968925,0.0017062408,0.0022303367,0.069832504,0.09171089,0.005382606,0.0041485648,0.7771697],"study_design_scores_gemma":[0.0012148705,0.010250319,0.14939322,0.00078695157,0.0007146876,0.005655562,0.0037293246,0.6147902,0.15623589,0.021086395,0.035652626,0.00049005076],"about_ca_topic_score_codex":0.00196876,"about_ca_topic_score_gemma":0.0032430626,"teacher_disagreement_score":0.011444819,"about_ca_system_score_codex":0.0009623643,"about_ca_system_score_gemma":0.001362463,"threshold_uncertainty_score":0.06052673},"labels":[],"label_agreement":null},{"id":"W1880492488","doi":"10.1120/jacmp.v14i1.4052","title":"Automated IMRT planning with regional optimization using planning scripts","year":2013,"lang":"en","type":"article","venue":"Journal of Applied Clinical Medical Physics","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":92,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"London Health Sciences Centre; Western University","funders":"Cancer Care Ontario","keywords":"Scripting language; Radiation treatment planning; Computer science; Medical physics; Medicine; Radiation therapy; Radiology; Programming language","score_opus":0.09085317414621082,"score_gpt":0.3731214052892368,"score_spread":0.282268231143026,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1880492488","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045113396,0.000054181088,0.9392417,0.000033097498,0.00004472108,0.00044507303,0.00070219155,0.050496396,0.0044713747],"genre_scores_gemma":[0.06572435,0.0000758742,0.91695374,0.00009497952,0.000023811574,0.0010469586,0.0020037612,0.009411376,0.004665265],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986652,0.00033637125,0.0001288578,0.00028050915,0.00048760689,0.00010148155],"domain_scores_gemma":[0.9978951,0.00090656354,0.00018534639,0.00030019885,0.000644133,0.00006854357],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014027053,0.001801383,0.0006308408,0.00075540924,0.00048612474,0.0012853992,0.0019833934,0.0006246355,0.02228814],"category_scores_gemma":[0.0033959781,0.00089436554,0.0007716178,0.00054651836,0.00044532795,0.00057912245,0.0007245499,0.0011512175,0.0054892064],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073142385,0.00026992548,0.0018455809,0.00075698755,0.00015054323,0.0004475575,0.0005363591,0.31282842,0.05920926,0.010708293,0.049223423,0.56329226],"study_design_scores_gemma":[0.0001233734,0.0001326033,0.0006746339,0.000056422596,0.000036079276,0.00016466463,0.000035464676,0.9010256,0.053980377,0.0031371848,0.04055973,0.00007387879],"about_ca_topic_score_codex":0.00247431,"about_ca_topic_score_gemma":0.0021992247,"teacher_disagreement_score":0.02228814,"about_ca_system_score_codex":0.0007132068,"about_ca_system_score_gemma":0.0014833404,"threshold_uncertainty_score":0.0745613},"labels":[],"label_agreement":null},{"id":"W1883222843","doi":"10.1109/cmpsac.1991.170177","title":"A new control flow representation","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"National Research Council Canada","keywords":"Representation (politics); Computer science; Control flow graph; Control flow; Flow (mathematics); Theoretical computer science; Control (management); Graph; Information flow; Control flow analysis; Programming language; Artificial intelligence; Mathematics; Procedural programming","score_opus":0.03387879740053827,"score_gpt":0.25941588629852164,"score_spread":0.22553708889798335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1883222843","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00071377424,0.000119634,0.99135274,0.00019524436,0.00014331924,0.000114104085,0.00040887855,0.0027406053,0.004211704],"genre_scores_gemma":[0.03499959,0.0004650582,0.94933844,0.0004979796,0.00022687468,0.00059104944,0.0022797496,0.0016794692,0.009921819],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9962352,0.0005787844,0.00040666567,0.00093123276,0.0015691364,0.00027901385],"domain_scores_gemma":[0.99701166,0.00089737965,0.00024148823,0.0009207413,0.0008114103,0.00011723289],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019752467,0.0015361827,0.00083614094,0.0030623043,0.0009004184,0.0043062665,0.0022112802,0.0016743365,0.015246316],"category_scores_gemma":[0.0056106616,0.0007436412,0.0019065122,0.0020681885,0.0019830887,0.0073294607,0.0022474749,0.0037587404,0.0039861305],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012434887,0.00011790081,0.00046760126,0.00043928457,0.000039764112,0.00033610495,0.00034911922,0.027081909,0.009953589,0.6951646,0.022977335,0.24294832],"study_design_scores_gemma":[0.00008778834,0.00014750233,0.00026949245,0.00026261603,0.00012486882,0.0007304921,0.00007794093,0.2062972,0.0143222585,0.44927284,0.3282905,0.000116480805],"about_ca_topic_score_codex":0.0033506672,"about_ca_topic_score_gemma":0.0022786842,"teacher_disagreement_score":0.015246316,"about_ca_system_score_codex":0.0013856296,"about_ca_system_score_gemma":0.0023049507,"threshold_uncertainty_score":0.051003933},"labels":[],"label_agreement":null},{"id":"W1898227670","doi":"10.1109/tools.1997.681887","title":"Object-Oriented Testing","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; System integration testing; White-box testing; Software reliability testing; Manual testing; Software engineering; Non-regression testing; Keyword-driven testing; Software performance testing; Integration testing; Black-box testing; Object-oriented programming; Test strategy; Software construction; Test Management Approach; Software; Class (philosophy); Software testing; System testing; Programming language; Software development; Artificial intelligence","score_opus":0.0277080684817965,"score_gpt":0.26312562346130897,"score_spread":0.23541755497951247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1898227670","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01718317,0.0025955322,0.77179897,0.001973241,0.0008619717,0.0006927217,0.00028844306,0.00634125,0.1982647],"genre_scores_gemma":[0.4587628,0.006511745,0.4169276,0.003335527,0.00095570687,0.0012884104,0.0019298788,0.0027974928,0.107490845],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.990012,0.0027204743,0.00041715908,0.0007740179,0.005555725,0.00052060187],"domain_scores_gemma":[0.98862004,0.0053855567,0.00053662003,0.0030755636,0.0019963672,0.00038584188],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005009614,0.0009950541,0.00067095837,0.0012632844,0.0006658559,0.0029700666,0.0023511807,0.0012171222,0.009566499],"category_scores_gemma":[0.017180348,0.00029901395,0.0005145236,0.0009644373,0.0016892103,0.0035350095,0.002516948,0.0012133955,0.0043331576],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013232001,0.00028455825,0.0031160188,0.00049712486,0.000052418316,0.0006251997,0.0009023368,0.0034900266,0.0109872725,0.36013007,0.023202885,0.5965798],"study_design_scores_gemma":[0.00015880667,0.00063638657,0.0026954338,0.0006659982,0.00007981547,0.003665585,0.00063092046,0.025459018,0.036037356,0.31729722,0.6125937,0.00007969271],"about_ca_topic_score_codex":0.0005024792,"about_ca_topic_score_gemma":0.00037767447,"teacher_disagreement_score":0.009566499,"about_ca_system_score_codex":0.0006579774,"about_ca_system_score_gemma":0.001045632,"threshold_uncertainty_score":0.032003105},"labels":[],"label_agreement":null},{"id":"W1900631367","doi":"10.1109/qrs-c.2015.26","title":"On the Effect of Counters in Guard Conditions When State-Based Multi-objective Testing","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Extended finite-state machine; Tree traversal; Guard (computer science); Finite-state machine; Computer science; Executable; Automaton; Graph; State (computer science); Deterministic finite automaton; Graph traversal; Theoretical computer science; Algorithm; Programming language","score_opus":0.047483217440556086,"score_gpt":0.2939482613152094,"score_spread":0.24646504387465332,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1900631367","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33481964,0.0007325231,0.65573376,0.0005704237,0.0001152554,0.0002110564,0.000068790134,0.0019031322,0.005845488],"genre_scores_gemma":[0.86885333,0.00010626909,0.13000634,0.00019028169,0.000023273911,0.000077012985,0.000042526797,0.00018368378,0.0005172338],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99016994,0.0060410425,0.0005272826,0.0007044811,0.002010296,0.000546909],"domain_scores_gemma":[0.8403934,0.14346361,0.0074156597,0.0054831863,0.0023261243,0.00091796933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077986685,0.0010752974,0.000629171,0.0010735148,0.00055001205,0.0016111417,0.0012852614,0.0013655897,0.0012137776],"category_scores_gemma":[0.07274357,0.00056039635,0.0005447746,0.0009270007,0.0018999594,0.0029216323,0.0012779173,0.0018364438,0.00016601758],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002385697,0.0005476791,0.02575014,0.0004406434,0.00016759611,0.0013835544,0.0012432607,0.74382395,0.035235602,0.03994143,0.00081337447,0.14826703],"study_design_scores_gemma":[0.00013562763,0.00071446574,0.0022947877,0.00011356996,0.00018048407,0.0004195966,0.00009095592,0.95122135,0.033397693,0.010001813,0.0013747545,0.000054859793],"about_ca_topic_score_codex":0.0023639314,"about_ca_topic_score_gemma":0.003389772,"teacher_disagreement_score":0.0077986685,"about_ca_system_score_codex":0.0009828326,"about_ca_system_score_gemma":0.0012816474,"threshold_uncertainty_score":0.04124379},"labels":[],"label_agreement":null},{"id":"W191325165","doi":"10.1007/978-3-642-31491-9_8","title":"Combining UML Sequence and State Machine Diagrams for Data-Flow Based Integration Testing","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Unified Modeling Language; Sequence diagram; Control flow; Finite-state machine; Applications of UML; Data flow diagram; Integration testing; Activity diagram; Abstract state machines; Graph; Data mining; Programming language; Class diagram; State (computer science); Theoretical computer science; Database; Software","score_opus":0.08609689577891642,"score_gpt":0.30227431284458683,"score_spread":0.2161774170656704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W191325165","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005059685,0.000112138936,0.99042636,0.0000502389,0.000026316713,0.00006344188,0.000063987405,0.0028861032,0.0013117406],"genre_scores_gemma":[0.10426741,0.00025147106,0.89200985,0.0000629917,0.000020186824,0.00020264064,0.0006113927,0.0010023727,0.0015717032],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99579287,0.0015163397,0.0003595322,0.00037958124,0.0017929851,0.0001586378],"domain_scores_gemma":[0.98994243,0.0067375,0.0004042727,0.0014015898,0.0013739669,0.00014018458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043543703,0.0013149757,0.00091589213,0.0033586302,0.000317215,0.0016897524,0.0015294192,0.0011168252,0.0037479284],"category_scores_gemma":[0.012577983,0.0009267596,0.001113218,0.0018914128,0.00058235414,0.0038451948,0.0018319974,0.0013088257,0.0011373846],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003251193,0.00040141537,0.0030826463,0.00075452344,0.00015037644,0.00049666315,0.0007513618,0.076392405,0.05397265,0.05350487,0.0050527533,0.8051153],"study_design_scores_gemma":[0.00009795774,0.00026155592,0.0013501708,0.00039024677,0.00023708004,0.0005964658,0.00015997911,0.85123086,0.050355054,0.07020188,0.025029132,0.00008947916],"about_ca_topic_score_codex":0.0017542704,"about_ca_topic_score_gemma":0.0024160463,"teacher_disagreement_score":0.0043543703,"about_ca_system_score_codex":0.000562046,"about_ca_system_score_gemma":0.00086875097,"threshold_uncertainty_score":0.023028374},"labels":[],"label_agreement":null},{"id":"W1916225420","doi":"10.1109/ictta.2004.1307898","title":"Timed test cases generation based on test purposes expressed as message sequence charts","year":2004,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Test (biology); Test case; Process (computing); Reliability engineering; Test Management Approach; Test suite; Fault (geology); Formal specification; Sequence (biology); Software engineering; Programming language; Software; Software system; Engineering; Software construction; Machine learning","score_opus":0.054010909486074114,"score_gpt":0.2868837883451243,"score_spread":0.23287287885905017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1916225420","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01866159,0.00009344606,0.9700726,0.00009979207,0.00006697142,0.0009288237,0.00047519634,0.0076210652,0.0019805373],"genre_scores_gemma":[0.4147584,0.0002597513,0.5771826,0.00014583977,0.0000831249,0.0018200486,0.0023262429,0.0008575457,0.0025664545],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956846,0.0019027814,0.00034080865,0.00045884043,0.0013670432,0.00024598744],"domain_scores_gemma":[0.9851327,0.010533193,0.0010942222,0.00091369636,0.0021244918,0.00020174088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002770752,0.002020945,0.0006863958,0.0033481761,0.00037693043,0.0014962066,0.0014687539,0.0010551694,0.004127013],"category_scores_gemma":[0.01912036,0.000440274,0.00094124733,0.0012339096,0.0009593515,0.0009921234,0.00066870585,0.0007978775,0.0010286459],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022593387,0.00067681045,0.0058837202,0.0010873937,0.00020093401,0.0035611277,0.0010811852,0.4317081,0.06034148,0.08666704,0.011542121,0.39499077],"study_design_scores_gemma":[0.00041013624,0.000536685,0.0008433703,0.00016115847,0.00011779099,0.0004845761,0.000091917515,0.89241225,0.06523913,0.030929303,0.0087027615,0.00007093892],"about_ca_topic_score_codex":0.0020503718,"about_ca_topic_score_gemma":0.001340083,"teacher_disagreement_score":0.004127013,"about_ca_system_score_codex":0.00079471455,"about_ca_system_score_gemma":0.0010717966,"threshold_uncertainty_score":0.014653325},"labels":[],"label_agreement":null},{"id":"W1920048997","doi":"10.1109/ase.1998.732614","title":"Testing using log file analysis: tools, methods, and issues","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":93,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada; University of British Columbia","keywords":"Computer science; Scope (computer science); Software; Software testing; Programming language; Database; Software engineering","score_opus":0.16286346450439254,"score_gpt":0.37333172560894495,"score_spread":0.2104682611045524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1920048997","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022101142,0.0056936727,0.98168683,0.005290472,0.0001186815,0.00013313479,0.00006933638,0.002608334,0.0021894188],"genre_scores_gemma":[0.07682523,0.010032353,0.90723324,0.0008851908,0.0006124445,0.00093356543,0.00019950097,0.0011679465,0.0021104394],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9475852,0.029442707,0.002884181,0.0030268966,0.01621464,0.0008462576],"domain_scores_gemma":[0.73554707,0.20886609,0.0067085396,0.032760635,0.013859474,0.0022581522],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04923257,0.0031706158,0.0025049485,0.009950952,0.0015808045,0.012199412,0.009852931,0.006032825,0.003500575],"category_scores_gemma":[0.13923626,0.0027428558,0.0014388676,0.008230467,0.015589237,0.032955512,0.0045427787,0.0066337306,0.0025556462],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014877487,0.0004250815,0.007809171,0.001575075,0.000105053216,0.0003391027,0.0017403957,0.00944681,0.0022213855,0.34936395,0.010041199,0.6167839],"study_design_scores_gemma":[0.00012589805,0.00030874836,0.0026752937,0.0021278104,0.00008194529,0.002355851,0.0013350585,0.15687384,0.008145825,0.7640285,0.061564144,0.00037714562],"about_ca_topic_score_codex":0.0028187567,"about_ca_topic_score_gemma":0.00134419,"teacher_disagreement_score":0.04923257,"about_ca_system_score_codex":0.0023016701,"about_ca_system_score_gemma":0.0038675405,"threshold_uncertainty_score":0.26036984},"labels":[],"label_agreement":null},{"id":"W1923117336","doi":"10.1007/978-3-642-10677-4_27","title":"Quasi-Deterministic Partially Observable Markov Decision Processes","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Observability; Computer science; Observable; Markov decision process; Hierarchy; Mathematical optimization; Theoretical computer science; Polynomial hierarchy; Decision problem; Time complexity; Markov process; Algorithm; Mathematics; Applied mathematics","score_opus":0.027048840545886504,"score_gpt":0.2727263273379048,"score_spread":0.2456774867920183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1923117336","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05002865,0.000864315,0.934983,0.001146582,0.00015162527,0.00009313128,0.0012225711,0.00037270243,0.011137329],"genre_scores_gemma":[0.882784,0.0016164689,0.0855514,0.00033384105,0.00022303304,0.00038097563,0.0013873618,0.00010554149,0.027617447],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99734604,0.001139368,0.00015293795,0.00048895733,0.0004907026,0.00038200468],"domain_scores_gemma":[0.9648758,0.030910764,0.0013260426,0.0010990129,0.0010942324,0.0006941095],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036176885,0.0011317827,0.0020693324,0.00073807436,0.00059120497,0.0028290302,0.0022180101,0.0021676878,0.012668254],"category_scores_gemma":[0.016042698,0.0011610216,0.0012591478,0.0011067733,0.0021830632,0.0032973816,0.0014201921,0.0024061874,0.0012170858],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024555495,0.0000685319,0.001394431,0.00015056423,0.00007294502,0.0002738159,0.00011518295,0.16573648,0.000757628,0.8200007,0.0017813302,0.009402922],"study_design_scores_gemma":[0.00008130411,0.000045831148,0.00038457994,0.00002146967,0.000027711416,0.00006314543,0.000015095834,0.6472093,0.00021413325,0.35106418,0.0008488889,0.00002447337],"about_ca_topic_score_codex":0.0043119215,"about_ca_topic_score_gemma":0.00359604,"teacher_disagreement_score":0.012668254,"about_ca_system_score_codex":0.0018972217,"about_ca_system_score_gemma":0.0017790123,"threshold_uncertainty_score":0.0423795},"labels":[],"label_agreement":null},{"id":"W1925282067","doi":"10.1109/icst.2015.7102588","title":"Exploring Test Suite Diversification and Code Coverage in Multi-Objective Test Case Selection","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Test suite; Heuristics; Computer science; Code coverage; Selection (genetic algorithm); Fault coverage; Test case; Suite; Code (set theory); Test (biology); Programming language; Machine learning; Software; Operating system; Engineering","score_opus":0.18669848225657762,"score_gpt":0.3023859569594612,"score_spread":0.11568747470288357,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1925282067","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5086362,0.0011518742,0.48460147,0.00041350303,0.000021786098,0.00019332317,0.00008455222,0.0005168083,0.004380588],"genre_scores_gemma":[0.92109245,0.00016130437,0.078003824,0.000081646365,0.000013228602,0.00011625944,0.00011758915,0.00005430895,0.00035949695],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963307,0.0023064383,0.00009644387,0.0002558209,0.0007682839,0.00024232],"domain_scores_gemma":[0.9869653,0.010744495,0.000914568,0.0004382892,0.0006905728,0.00024682918],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004197434,0.0013801404,0.0009998793,0.0029222174,0.00035482683,0.00097153726,0.0011832933,0.0009302584,0.00074293365],"category_scores_gemma":[0.013724979,0.00050636724,0.0008652345,0.00126136,0.000836429,0.0011435895,0.0011114188,0.0006399427,0.00011320155],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020955707,0.00023656101,0.014026982,0.00015622479,0.00021853128,0.00025133038,0.00018618388,0.8832883,0.00634983,0.0035127306,0.00040018762,0.09116356],"study_design_scores_gemma":[0.00003278205,0.00021338127,0.002547374,0.000025275223,0.000048873764,0.0001013343,0.000052642386,0.99199075,0.0024789406,0.0021339846,0.0003639738,0.000010647758],"about_ca_topic_score_codex":0.0018229183,"about_ca_topic_score_gemma":0.0020942658,"teacher_disagreement_score":0.004197434,"about_ca_system_score_codex":0.0009258174,"about_ca_system_score_gemma":0.0011853753,"threshold_uncertainty_score":0.022198439},"labels":[],"label_agreement":null},{"id":"W1928770023","doi":"10.1109/icsm.1989.65194","title":"Insights into regression testing (software testing)","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":184,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Regression testing; Testability; Test plan; Computer science; Non-regression testing; Test (biology); Plan (archaeology); Risk-based testing; Regression analysis; Test Management Approach; Manual testing; Test strategy; Set (abstract data type); Reliability engineering; Regression; Keyword-driven testing; White-box testing; Machine learning; Data mining; Measure (data warehouse); Software; Statistics; Programming language; Mathematics; Engineering; Software system","score_opus":0.04500660842911066,"score_gpt":0.27565002333236094,"score_spread":0.23064341490325027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1928770023","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0076943752,0.013765207,0.90367734,0.011028845,0.00037991424,0.00009022399,0.00017227346,0.0007150738,0.06247678],"genre_scores_gemma":[0.40133238,0.020328278,0.54469407,0.005422507,0.0016316704,0.00043264971,0.00047375198,0.00054850994,0.025136195],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9953956,0.002393882,0.00019061363,0.0004382765,0.0012955954,0.00028604316],"domain_scores_gemma":[0.98897165,0.008752327,0.0005513321,0.00056223373,0.0009491758,0.00021338166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034388832,0.0015811552,0.00081550586,0.0034823618,0.0006659047,0.003657505,0.0014749421,0.0020422428,0.005940976],"category_scores_gemma":[0.01493326,0.00070875697,0.0010635823,0.0025718783,0.005660327,0.0058906386,0.0016506857,0.0037579334,0.0013488805],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025038236,0.000064704625,0.0008292695,0.00025163783,0.000017997307,0.00028545182,0.0007070996,0.008701543,0.00061752566,0.9049387,0.005502894,0.078058116],"study_design_scores_gemma":[0.00001423948,0.000054448352,0.000762147,0.00018209542,0.000013876115,0.0004448132,0.00019364538,0.024171757,0.00046652535,0.9359081,0.0377664,0.00002195712],"about_ca_topic_score_codex":0.0038507623,"about_ca_topic_score_gemma":0.0023093468,"teacher_disagreement_score":0.005940976,"about_ca_system_score_codex":0.002309652,"about_ca_system_score_gemma":0.0012625614,"threshold_uncertainty_score":0.019874513},"labels":[],"label_agreement":null},{"id":"W1947217269","doi":"10.1109/ase.1998.732610","title":"Programmatic testing of the Standard Template Library containers","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Test suite; Computer science; White-box testing; Conformance testing; Suite; Unit testing; Software engineering; Programming language; Code coverage; Key (lock); Keyword-driven testing; Code (set theory); Black box; Component (thermodynamics); Test case; Software; Operating system; Software development; Software construction; Artificial intelligence; Standardization; Machine learning; Set (abstract data type)","score_opus":0.037983183024864135,"score_gpt":0.22401962082305504,"score_spread":0.1860364377981909,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1947217269","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24174888,0.00012032199,0.7284021,0.00035353642,0.000032980017,0.00029460582,0.00032303552,0.022414705,0.0063099405],"genre_scores_gemma":[0.7515819,0.000084260544,0.2413493,0.00020796373,0.00002225499,0.00027156726,0.000806037,0.0026133137,0.00306332],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9964154,0.0011308584,0.00022911177,0.00042694496,0.001493297,0.00030439196],"domain_scores_gemma":[0.989645,0.005356614,0.00088799093,0.0026508388,0.0011867732,0.00027270662],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002688879,0.0007104307,0.00039191943,0.0005774742,0.0003655541,0.0010890436,0.0019165685,0.00072724465,0.0031029093],"category_scores_gemma":[0.0133044,0.000446165,0.0005421439,0.0006397848,0.0016392665,0.0021668784,0.0012220627,0.0009137452,0.0008263612],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012986647,0.0013189592,0.026019936,0.0006641032,0.000118169344,0.0032578649,0.0017297803,0.11652402,0.24229102,0.08085349,0.016664093,0.50926],"study_design_scores_gemma":[0.00022094445,0.001102696,0.0052517587,0.00007419445,0.000061279556,0.0017188472,0.00017120823,0.5407841,0.39480585,0.033159256,0.0225569,0.00009298006],"about_ca_topic_score_codex":0.0015598744,"about_ca_topic_score_gemma":0.0013956884,"teacher_disagreement_score":0.0031029093,"about_ca_system_score_codex":0.0008386388,"about_ca_system_score_gemma":0.0011877278,"threshold_uncertainty_score":0.014220357},"labels":[],"label_agreement":null},{"id":"W1952910495","doi":"10.1109/enabl.1999.805197","title":"Static analysis of binary code to isolate malicious behaviors","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Executable; Program slicing; Static program analysis; Static analysis; Programming language; Slicing; Program analysis; Code (set theory); Binary code; Semantics (computer science); Source code; Binary number; Software; Software development","score_opus":0.02127857445463315,"score_gpt":0.29714079144376576,"score_spread":0.2758622169891326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1952910495","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31076306,0.00055641995,0.6730073,0.00024855108,0.000064478256,0.00020428476,0.00022094198,0.009083454,0.005851539],"genre_scores_gemma":[0.75468254,0.00026077917,0.24159782,0.000098310746,0.000027561322,0.00008680196,0.00036666408,0.00072122784,0.0021583242],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993735,0.000112912865,0.00004293752,0.00009358521,0.00031226798,0.000064847394],"domain_scores_gemma":[0.9957489,0.0015056751,0.00095359184,0.00085684774,0.00083074556,0.00010428074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000562818,0.0006415558,0.00047386764,0.001976613,0.0005663532,0.00059818197,0.00047482125,0.00041603734,0.0019006929],"category_scores_gemma":[0.0036358882,0.00035191202,0.0004642347,0.00073404843,0.0010148968,0.00094546995,0.00045015634,0.00049308396,0.00047337645],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005061023,0.0003165652,0.029277781,0.00068780687,0.00008119729,0.0012448335,0.00102785,0.07534814,0.51355416,0.03644045,0.0024428773,0.3390723],"study_design_scores_gemma":[0.000033087937,0.0003626827,0.012393778,0.00012301379,0.000095075906,0.0009771277,0.00013730288,0.6543509,0.30292726,0.02174037,0.0067834193,0.000076023956],"about_ca_topic_score_codex":0.002612572,"about_ca_topic_score_gemma":0.003167275,"teacher_disagreement_score":0.002612572,"about_ca_system_score_codex":0.0005058265,"about_ca_system_score_gemma":0.00092598546,"threshold_uncertainty_score":0.0063585043},"labels":[],"label_agreement":null},{"id":"W1953128469","doi":"10.1109/scam.2001.972662","title":"A hybrid program slicing framework","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Program slicing; Slicing; Computer science; Program comprehension; Executable; Debugging; Programming language; Software maintenance; Software; Object-oriented programming; Software engineering; Software system","score_opus":0.032270738660680844,"score_gpt":0.276433132802871,"score_spread":0.24416239414219018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1953128469","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018913569,0.00011300304,0.99548495,0.00005164344,0.000005147584,0.000038301107,0.00004147859,0.0011030971,0.0012709424],"genre_scores_gemma":[0.061061658,0.00030354105,0.93626755,0.00006882475,0.000019412948,0.00013597767,0.00018374421,0.00027366375,0.0016856766],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991744,0.00019264527,0.00005317821,0.00014942998,0.00035725336,0.000073185125],"domain_scores_gemma":[0.9991371,0.00032825797,0.000064642765,0.00020567598,0.00022307763,0.00004126899],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014280105,0.00067888014,0.0004958634,0.00094592315,0.00043712978,0.0010068944,0.0013940278,0.0006216582,0.0026805422],"category_scores_gemma":[0.0017891864,0.00039660148,0.0009218718,0.0006575217,0.0014549902,0.0021286772,0.0010770566,0.00092099654,0.00047124265],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011189477,0.000068635854,0.001159237,0.00038066393,0.00009036109,0.00044355678,0.0009388615,0.107940406,0.028032327,0.5604371,0.0049584825,0.29543856],"study_design_scores_gemma":[0.000063179556,0.00012499429,0.0005406511,0.00013915478,0.00010177733,0.00058967323,0.00015038444,0.63261086,0.019224247,0.26835114,0.078041166,0.000062800624],"about_ca_topic_score_codex":0.0041336874,"about_ca_topic_score_gemma":0.00548879,"teacher_disagreement_score":0.0041336874,"about_ca_system_score_codex":0.0007750017,"about_ca_system_score_gemma":0.00145847,"threshold_uncertainty_score":0.00896728},"labels":[],"label_agreement":null},{"id":"W1956656014","doi":"10.1007/3-540-44830-6_15","title":"Fault Diagnosis in Extended Finite State Machines","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Finite-state machine; Computer science; Test suite; State (computer science); Suite; Fault (geology); Finite state; Algorithm; Test case; Machine learning","score_opus":0.020160453791129414,"score_gpt":0.26465815315733293,"score_spread":0.24449769936620352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1956656014","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02155762,0.0006095934,0.9746606,0.0001047904,0.000058117872,0.000022528611,0.000043831573,0.0010313578,0.0019116962],"genre_scores_gemma":[0.6692845,0.0006938161,0.32471025,0.0001099435,0.00008410285,0.000078127065,0.00026865726,0.00015199285,0.004618567],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990402,0.00025187054,0.00008050066,0.00019243543,0.00033606574,0.00009894265],"domain_scores_gemma":[0.99453247,0.0042245598,0.00021999715,0.0005724972,0.00038855642,0.00006186343],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001071955,0.0005531555,0.00079987774,0.00078047864,0.00032650054,0.0008322535,0.0012019932,0.0009066132,0.0024802478],"category_scores_gemma":[0.006440488,0.00051444146,0.00081672915,0.0006404879,0.0013222799,0.0031535309,0.0009128024,0.0015523126,0.00029895836],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006363058,0.00012175737,0.0020818457,0.0006434998,0.00012699982,0.00095181353,0.0006758212,0.3799777,0.017174758,0.22591013,0.0033044245,0.36839494],"study_design_scores_gemma":[0.000038868635,0.00008517889,0.000455013,0.0000924489,0.00004096148,0.0002721275,0.000037826714,0.6829015,0.011730314,0.30113587,0.0031869193,0.000022968414],"about_ca_topic_score_codex":0.001041569,"about_ca_topic_score_gemma":0.0009942874,"teacher_disagreement_score":0.0024802478,"about_ca_system_score_codex":0.00064166496,"about_ca_system_score_gemma":0.0005041723,"threshold_uncertainty_score":0.008297205},"labels":[],"label_agreement":null},{"id":"W1959797194","doi":"10.1007/978-3-642-31057-7_30","title":"Application-Only Call Graph Construction","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":100,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Call graph; Computer science; Graph; Theoretical computer science; Programming language","score_opus":0.013995644423430504,"score_gpt":0.24472505283389454,"score_spread":0.23072940841046402,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1959797194","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010282399,0.00015669428,0.93669474,0.00030235254,0.00020870812,0.0002476018,0.00041176256,0.020883093,0.030812636],"genre_scores_gemma":[0.2515215,0.00040121365,0.6940723,0.00061894965,0.0001333429,0.00037795096,0.0026801429,0.009512159,0.040682495],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984871,0.00023852113,0.00007052082,0.00029816706,0.00065635145,0.0002493727],"domain_scores_gemma":[0.99669826,0.0006410283,0.000084692794,0.0019048111,0.000542235,0.0001289464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007954176,0.0009853579,0.0008607149,0.001564376,0.0009733699,0.0016419832,0.0021626346,0.0011670402,0.022659468],"category_scores_gemma":[0.00418896,0.00097164465,0.0013177709,0.0013418287,0.0011590498,0.002925879,0.003744458,0.0029071802,0.010296748],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025456707,0.00037520606,0.0018029083,0.0007564115,0.000108593995,0.0005929249,0.0004748183,0.010827838,0.052434053,0.19179223,0.055589832,0.6849906],"study_design_scores_gemma":[0.00010344329,0.00024887442,0.0028208848,0.00025317664,0.0003403493,0.0022084417,0.0002858353,0.17876962,0.15165572,0.36937213,0.2937928,0.0001487637],"about_ca_topic_score_codex":0.0012834362,"about_ca_topic_score_gemma":0.0024906218,"teacher_disagreement_score":0.022659468,"about_ca_system_score_codex":0.00051161717,"about_ca_system_score_gemma":0.002039242,"threshold_uncertainty_score":0.07580352},"labels":[],"label_agreement":null},{"id":"W1962250603","doi":"10.1109/esem.2015.7321218","title":"Will This Bug-Fixing Change Break Regression Testing?","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Regression testing; Computer science; Regression analysis; Machine learning; Programming language; Software; Software development","score_opus":0.15915339063956024,"score_gpt":0.3114980059185787,"score_spread":0.15234461527901844,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1962250603","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9080257,0.003148154,0.072171,0.0024687082,0.00045310849,0.00024014032,0.0035094281,0.006042219,0.003941664],"genre_scores_gemma":[0.9738895,0.0006585484,0.021470034,0.00037386327,0.00012154964,0.00004649479,0.0021331515,0.0002224712,0.0010844744],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.99711883,0.00033561443,0.00026789203,0.0009868835,0.0010515068,0.00023933231],"domain_scores_gemma":[0.98159045,0.007955152,0.006151481,0.0011501026,0.0026843504,0.00046849262],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019717305,0.0010269877,0.0005477585,0.0030803122,0.0005223565,0.0011040648,0.00077212084,0.0012197172,0.0013292251],"category_scores_gemma":[0.029840225,0.0003422352,0.00085257355,0.0017841291,0.00045616785,0.0018971224,0.0005591894,0.0010586758,0.0007192639],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024366118,0.00015717358,0.70059353,0.00045506665,0.00021269194,0.0016731048,0.00056291494,0.007400697,0.008531556,0.0005783987,0.0052863397,0.27430487],"study_design_scores_gemma":[0.000039405775,0.0005440741,0.80847627,0.0004707359,0.0005564842,0.00646838,0.0010730198,0.1387912,0.02251531,0.004313991,0.016637783,0.00011332007],"about_ca_topic_score_codex":0.006642069,"about_ca_topic_score_gemma":0.009463177,"teacher_disagreement_score":0.006642069,"about_ca_system_score_codex":0.0006319905,"about_ca_system_score_gemma":0.00068293884,"threshold_uncertainty_score":0.01320684},"labels":[],"label_agreement":null},{"id":"W1963583462","doi":"10.1109/issre.2012.20","title":"Mutation Testing of Event Processing Queries","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Event (particle physics); Benchmark (surveying); Complex event processing; Mutation; Data mining; Process (computing); SQL; Quality (philosophy); Mutation testing; Database; Programming language","score_opus":0.0418201916147039,"score_gpt":0.29719331305155433,"score_spread":0.2553731214368504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1963583462","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46031064,0.00013167544,0.5286319,0.0002588516,0.000038187525,0.0004363089,0.00034657234,0.007780785,0.0020650758],"genre_scores_gemma":[0.865597,0.000067846944,0.13226876,0.00019801204,0.000011372798,0.00017981944,0.0005481342,0.000354795,0.0007743027],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99395776,0.001657277,0.0005202053,0.0009770949,0.0024678158,0.00041993154],"domain_scores_gemma":[0.98667496,0.008315702,0.001450273,0.0014613118,0.001844342,0.00025340062],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029479617,0.0010075927,0.0004808891,0.0011234542,0.00036090644,0.0009938385,0.0017002003,0.0008693545,0.0008378844],"category_scores_gemma":[0.016430166,0.00026805545,0.00094349106,0.0005847433,0.0013004754,0.0016154099,0.0010901567,0.00079799467,0.00018580098],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014983137,0.0011763136,0.04928519,0.00075429666,0.00022630692,0.0041969125,0.0010541426,0.2366173,0.42591748,0.026176788,0.0033500055,0.24974693],"study_design_scores_gemma":[0.00008908258,0.0007187322,0.0038251767,0.000044234326,0.00006993086,0.0010418718,0.00015266424,0.7611229,0.22340603,0.0064407415,0.0030373496,0.00005123345],"about_ca_topic_score_codex":0.0011878998,"about_ca_topic_score_gemma":0.00076412415,"teacher_disagreement_score":0.0029479617,"about_ca_system_score_codex":0.0007967379,"about_ca_system_score_gemma":0.001055831,"threshold_uncertainty_score":0.015590489},"labels":[],"label_agreement":null},{"id":"W1963871921","doi":"10.1145/2532352.2532353","title":"Document driven certification of computational science and engineering software","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"McMaster University","keywords":"Software engineering; Computer science; Documentation; Certification; Programmer; Programming language; Traceability; Software development; Verification and validation; Software quality; Software construction; Software requirements specification; Software; Engineering","score_opus":0.013694360057219442,"score_gpt":0.23651953913174364,"score_spread":0.2228251790745242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1963871921","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013458718,0.000112833244,0.9782588,0.00039111855,0.00006556002,0.00042902993,0.000091677764,0.0028450494,0.004347196],"genre_scores_gemma":[0.09717718,0.00022974427,0.89519835,0.00020032843,0.000039889313,0.0006344398,0.00071663776,0.0013152319,0.0044882954],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.97332925,0.010677111,0.0024963699,0.0019218829,0.010877881,0.00069746387],"domain_scores_gemma":[0.9151084,0.028041475,0.005425209,0.026903901,0.023479834,0.0010411664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01954675,0.00065590226,0.0005752413,0.0032288593,0.0012673982,0.0038012606,0.0029803286,0.001877747,0.0026897078],"category_scores_gemma":[0.07091771,0.00092395936,0.00078586774,0.0016444363,0.002387006,0.0043148682,0.004035556,0.0031472684,0.001665418],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023944829,0.00065418304,0.0056029335,0.0012865454,0.00006419746,0.0011270707,0.007329851,0.03179371,0.03593139,0.20515148,0.010366204,0.700453],"study_design_scores_gemma":[0.00034703413,0.0011006783,0.00445422,0.0018552871,0.00012898144,0.0034403596,0.0015953806,0.28492603,0.17189208,0.13307276,0.39682946,0.00035780595],"about_ca_topic_score_codex":0.0019366762,"about_ca_topic_score_gemma":0.0021207803,"teacher_disagreement_score":0.01954675,"about_ca_system_score_codex":0.0020688027,"about_ca_system_score_gemma":0.0063601458,"threshold_uncertainty_score":0.1033743},"labels":[],"label_agreement":null},{"id":"W1965117069","doi":"10.1145/2070336.2070357","title":"Enhancing spark's contract checking facilities using symbolic execution","year":2011,"lang":"en","type":"article","venue":"ACM SIGAda Ada Letters","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Symbolic execution; SPARK (programming language); Software engineering; Usability; Programming language; Automation; Design by contract; Software; Model checking; Formal methods; Software development; Software construction; Operating system; Engineering","score_opus":0.06601541650421534,"score_gpt":0.25624669602452055,"score_spread":0.19023127952030522,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1965117069","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02147941,0.00012040576,0.94097126,0.0002842129,0.00006879874,0.000102001075,0.0001364095,0.032002736,0.0048347902],"genre_scores_gemma":[0.27367875,0.00023214881,0.71661955,0.00020261438,0.00005542487,0.00013399453,0.0005730777,0.00447629,0.004028204],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.992785,0.0018384298,0.00047785416,0.0007611186,0.0036186317,0.00051892595],"domain_scores_gemma":[0.9766451,0.01327338,0.0013379819,0.0053709764,0.0029691376,0.0004033564],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057980446,0.0010401326,0.00090729154,0.0019459483,0.00085447344,0.0019676923,0.0028112275,0.00097394217,0.0044859564],"category_scores_gemma":[0.02074078,0.0009568507,0.0013538973,0.0011729264,0.0027758637,0.003818526,0.0031448533,0.0025072836,0.0014441905],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014495163,0.00060637627,0.010889454,0.00088575797,0.00023101388,0.0012937405,0.002118909,0.12908304,0.07472716,0.18026678,0.023693953,0.5747543],"study_design_scores_gemma":[0.00042773888,0.00034100172,0.0012102169,0.00014552611,0.00011078256,0.0009492374,0.00019224548,0.79042006,0.096331574,0.05898916,0.050718702,0.0001636984],"about_ca_topic_score_codex":0.0055436464,"about_ca_topic_score_gemma":0.005620338,"teacher_disagreement_score":0.0057980446,"about_ca_system_score_codex":0.0011234949,"about_ca_system_score_gemma":0.003959149,"threshold_uncertainty_score":0.030663371},"labels":[],"label_agreement":null},{"id":"W1967187747","doi":"10.1007/s10515-011-0093-0","title":"Prioritizing test cases with string distances","year":2011,"lang":"en","type":"article","venue":"Automated Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":125,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Test suite; String (physics); Computer science; Random testing; Code (set theory); Test case; Algorithm; Property (philosophy); Suite; Test (biology); Data mining; Mathematics; Machine learning; Programming language","score_opus":0.019894860361199286,"score_gpt":0.21321092719486978,"score_spread":0.1933160668336705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1967187747","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19645113,0.0008827432,0.7877273,0.00058685505,0.0001978494,0.0005529489,0.00052777404,0.007391446,0.0056819976],"genre_scores_gemma":[0.50954896,0.00023159522,0.4851899,0.00025438028,0.000101810925,0.00023506572,0.0012007081,0.00077527575,0.0024622828],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9866136,0.00466725,0.0010198883,0.0013332135,0.0055875042,0.0007785946],"domain_scores_gemma":[0.95046246,0.038252234,0.0024464002,0.002701787,0.0047918228,0.0013452553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004832741,0.0022838644,0.0021470983,0.007902329,0.00065729243,0.0021627469,0.0028615107,0.001696307,0.0050370456],"category_scores_gemma":[0.041589238,0.00075939845,0.001316186,0.0035347238,0.0010369824,0.0036615506,0.0025509296,0.0015757551,0.0009981827],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003220589,0.0009394062,0.020231348,0.0012529724,0.0003200633,0.0013428356,0.00033477886,0.11432973,0.054256238,0.026037233,0.0057919924,0.77194285],"study_design_scores_gemma":[0.00031203954,0.0011823287,0.004474725,0.00016470684,0.00030975876,0.00094136357,0.00026592973,0.86315995,0.059773996,0.06433971,0.004987905,0.00008749074],"about_ca_topic_score_codex":0.002420063,"about_ca_topic_score_gemma":0.0044715796,"teacher_disagreement_score":0.007902329,"about_ca_system_score_codex":0.0012197573,"about_ca_system_score_gemma":0.0024833656,"threshold_uncertainty_score":0.025558293},"labels":[],"label_agreement":null},{"id":"W1967729292","doi":"10.1145/1541822.1541824","title":"Cookies","year":2009,"lang":"en","type":"article","venue":"ACM Transactions on the Web","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Software deployment; Web testing; Web application security; The Internet; Web engineering; World Wide Web; Web application; Software engineering; Test strategy; Web analytics; Software; Web development; Computer security; Programming language","score_opus":0.028206095281323475,"score_gpt":0.2571300410262274,"score_spread":0.2289239457449039,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1967729292","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5684017,0.0038186673,0.117838375,0.0016426876,0.0005807657,0.002160341,0.011721093,0.018090785,0.27574563],"genre_scores_gemma":[0.73897076,0.0021884309,0.09623203,0.00079024304,0.000080625425,0.00060783426,0.010164093,0.003113315,0.14785266],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969292,0.0004886183,0.00020372623,0.0004990429,0.0015868923,0.00029248215],"domain_scores_gemma":[0.98107165,0.008481548,0.0011956285,0.0053467816,0.0034261018,0.00047827058],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002161899,0.0007037809,0.0004892036,0.004163507,0.0016644907,0.002515576,0.0012036843,0.0010777331,0.03582576],"category_scores_gemma":[0.017353931,0.0005154703,0.0005485815,0.003870084,0.0010559708,0.004407771,0.0021669338,0.0012178128,0.009022931],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020673366,0.00091891695,0.068654425,0.0020226992,0.00016716744,0.0032304907,0.006140086,0.0051648603,0.019659288,0.08253447,0.054458413,0.7549819],"study_design_scores_gemma":[0.00012862711,0.0012756755,0.083759144,0.0009216201,0.00025766072,0.01069987,0.0051137144,0.015953988,0.046207927,0.02778874,0.80756164,0.00033138765],"about_ca_topic_score_codex":0.0028564513,"about_ca_topic_score_gemma":0.004881524,"teacher_disagreement_score":0.03582576,"about_ca_system_score_codex":0.0008622018,"about_ca_system_score_gemma":0.0013016274,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W1968023441","doi":"10.5555/2818754.2818831","title":"DASE: document-assisted symbolic execution for improving automated software testing","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Symbolic execution; Computer science; Documentation; Heuristics; Software; Programming language; Focus (optics); Software bug; Static analysis; Software engineering; Data mining; Operating system","score_opus":0.06403600449072984,"score_gpt":0.30777833418900413,"score_spread":0.2437423296982743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1968023441","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04506927,0.00045303008,0.9112162,0.00035383104,0.00006531274,0.00018462274,0.0006257195,0.03983953,0.0021925548],"genre_scores_gemma":[0.19717216,0.00017956553,0.7986135,0.0001560308,0.000030985524,0.00015269409,0.0014128543,0.0011842771,0.0010978908],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99545187,0.0016606739,0.00039222697,0.0005063004,0.0017886146,0.00020036355],"domain_scores_gemma":[0.9859071,0.0073978007,0.0013870122,0.0026936722,0.0023914282,0.00022301119],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002561145,0.0013194035,0.00076153985,0.0025100855,0.00037864078,0.0013526228,0.0019471843,0.0007793709,0.0038710325],"category_scores_gemma":[0.012771219,0.00048095672,0.0007314776,0.001519012,0.0009541766,0.001908295,0.0016572034,0.0013029185,0.0011804085],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005466166,0.0005286836,0.00874268,0.0006400116,0.0001484288,0.00030627634,0.00032279184,0.09126619,0.07781721,0.008993716,0.013072803,0.79761463],"study_design_scores_gemma":[0.00020269888,0.00027683098,0.0015261422,0.00007483983,0.000056273435,0.00025523035,0.0000747233,0.890065,0.09121714,0.005521331,0.0106633995,0.00006650161],"about_ca_topic_score_codex":0.0036235217,"about_ca_topic_score_gemma":0.00519935,"teacher_disagreement_score":0.0038710325,"about_ca_system_score_codex":0.00076527474,"about_ca_system_score_gemma":0.0020410013,"threshold_uncertainty_score":0.0135448575},"labels":[],"label_agreement":null},{"id":"W1969019712","doi":"10.5555/1899721.1899863","title":"Managing verification error traces with bounded model debugging","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Vennsa Technologies (Canada)","funders":"","keywords":"Debugging; Debugger; Computer science; Bounded function; Model checking; Key (lock); Programming language; Theoretical computer science; Computer engineering; Operating system","score_opus":0.021675010745560185,"score_gpt":0.2617063323891501,"score_spread":0.2400313216435899,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1969019712","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030604944,0.0001282164,0.96469617,0.00028365405,0.00002173274,0.000044911543,0.000024642728,0.0035061256,0.00068960513],"genre_scores_gemma":[0.6861656,0.00011280419,0.31226188,0.00013113537,0.00002534436,0.00010205774,0.00011482278,0.00032810023,0.0007582295],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99304295,0.0040270104,0.00031829657,0.00058238086,0.0015206259,0.0005088247],"domain_scores_gemma":[0.96496916,0.021445544,0.0022011565,0.008831313,0.0019725375,0.0005802217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00541111,0.0011180711,0.0009432345,0.0010969181,0.00084727997,0.0017436923,0.0029498897,0.0014592739,0.0023859334],"category_scores_gemma":[0.031443115,0.0005988294,0.000717042,0.00072799594,0.0019226244,0.0050785188,0.004329473,0.001994444,0.00054751476],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000686198,0.0005069425,0.009834213,0.00022134473,0.00010110847,0.00045062994,0.00069316995,0.60044706,0.028880557,0.07167384,0.0032784198,0.28322652],"study_design_scores_gemma":[0.000030567804,0.000115933326,0.00016210791,0.00002876398,0.000017808732,0.00012453852,0.000044446424,0.9470518,0.017875165,0.03340914,0.0011184672,0.000021201786],"about_ca_topic_score_codex":0.0023586068,"about_ca_topic_score_gemma":0.0023647298,"teacher_disagreement_score":0.00541111,"about_ca_system_score_codex":0.0009889558,"about_ca_system_score_gemma":0.0026496311,"threshold_uncertainty_score":0.028616965},"labels":[],"label_agreement":null},{"id":"W1969596141","doi":"10.3166/jesa.43.889-904","title":"Test exhaustif de contrôleurs logiques spécifiés en Grafcet Apports et limites d'une modélisation par machines de Mealy","year":2009,"lang":"fr","type":"article","venue":"Journal Européen des Systèmes Automatisés","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Agence Nationale de la Recherche","keywords":"Mathematics","score_opus":0.03647001407798449,"score_gpt":0.30049436481604463,"score_spread":0.26402435073806013,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1969596141","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17447902,0.00012507677,0.81647754,0.00015798178,0.00005244636,0.00023441522,0.00020299872,0.0053425007,0.0029280158],"genre_scores_gemma":[0.6801937,0.00013775201,0.31200358,0.00015570513,0.00002410439,0.0006927003,0.0006080672,0.0010795383,0.0051048],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9916346,0.0022291017,0.0005252293,0.0013452797,0.0037389735,0.0005268462],"domain_scores_gemma":[0.9746943,0.0154189095,0.0015831843,0.0051374873,0.0028673818,0.0002986173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005171672,0.0009800139,0.0011081098,0.0011403641,0.00063805265,0.002090971,0.0017473067,0.0014672902,0.0038666278],"category_scores_gemma":[0.026030548,0.00068033644,0.0014022145,0.00043955643,0.0028074165,0.003035684,0.0018885848,0.002057584,0.00058185396],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005304,0.00081836985,0.033348516,0.0018384861,0.00044987575,0.0028493148,0.0043680347,0.2951803,0.26012236,0.11948731,0.0033035362,0.27292988],"study_design_scores_gemma":[0.00034285482,0.0019335842,0.0067820647,0.00021833078,0.00015370386,0.0010322948,0.00046294343,0.6055836,0.33760604,0.029403279,0.016284816,0.0001965797],"about_ca_topic_score_codex":0.0037580868,"about_ca_topic_score_gemma":0.0031532627,"teacher_disagreement_score":0.005171672,"about_ca_system_score_codex":0.0012826908,"about_ca_system_score_gemma":0.0018469942,"threshold_uncertainty_score":0.027350724},"labels":[],"label_agreement":null},{"id":"W1971159340","doi":"10.1145/2534397","title":"Test compaction techniques for assertion-based test generation","year":2013,"lang":"en","type":"article","venue":"ACM Transactions on Design Automation of Electronic Systems","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; McGill University","funders":"","keywords":"Assertion; Computer science; Compaction; Test (biology); Automatic test pattern generation; Path (computing); Scheme (mathematics); Reduction (mathematics); Test case; Cluster analysis; Algorithm; Programming language; Artificial intelligence; Machine learning; Mathematics","score_opus":0.03796342581928601,"score_gpt":0.272911710833078,"score_spread":0.234948285013792,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1971159340","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013434651,0.00013957293,0.98122,0.000070222864,0.000023722298,0.00013885216,0.00010207646,0.003998903,0.00087190385],"genre_scores_gemma":[0.32380772,0.00019631934,0.672349,0.00018566873,0.000052693904,0.00041627086,0.0006675885,0.00089909445,0.0014256836],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9960771,0.0013193026,0.00041678658,0.00040745473,0.0015785376,0.00020076722],"domain_scores_gemma":[0.98291415,0.010089993,0.0017054474,0.00346316,0.001675047,0.00015216455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022699838,0.00094851514,0.0006675527,0.0018633788,0.00039913587,0.0008629145,0.0014201947,0.0007075138,0.0043084444],"category_scores_gemma":[0.015478094,0.0005327338,0.00079745,0.0014792898,0.0009236544,0.0014843802,0.0012550033,0.000971409,0.0010642974],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009110226,0.0003498535,0.0037068285,0.00071877276,0.00014979746,0.0011832593,0.0007617172,0.12526734,0.1531346,0.044624586,0.005335247,0.663857],"study_design_scores_gemma":[0.00021354444,0.00078409316,0.002456336,0.00015488024,0.00016623578,0.0012985053,0.00015520597,0.7704379,0.17853554,0.031158552,0.014560146,0.00007904129],"about_ca_topic_score_codex":0.00083952956,"about_ca_topic_score_gemma":0.0009759738,"teacher_disagreement_score":0.0043084444,"about_ca_system_score_codex":0.0005710263,"about_ca_system_score_gemma":0.0006038521,"threshold_uncertainty_score":0.014413118},"labels":[],"label_agreement":null},{"id":"W1973085420","doi":"10.1145/1858996.1859079","title":"Search-carrying code","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Certification; Code (set theory); Programming language; Redundant code; State (computer science); Dead code; Code generation; Operating system; Key (lock)","score_opus":0.029951446062903547,"score_gpt":0.2906130173078692,"score_spread":0.26066157124496564,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1973085420","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0075696986,0.00019957626,0.97128254,0.00048439088,0.0001858989,0.0002370913,0.00024323186,0.009525515,0.010272007],"genre_scores_gemma":[0.2768021,0.00069460284,0.6943675,0.0010237925,0.00016840854,0.00058150565,0.0012157037,0.004514704,0.020631656],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9958735,0.0008388542,0.00026277607,0.0006483613,0.0019311169,0.00044537848],"domain_scores_gemma":[0.9888737,0.0041471925,0.0005859614,0.0041822414,0.0019016588,0.00030918233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030180642,0.0009519468,0.00079032715,0.0012994987,0.001158596,0.0020728805,0.003071349,0.0013591754,0.011401531],"category_scores_gemma":[0.013478607,0.0006642267,0.001598741,0.0009680815,0.003995463,0.0047541936,0.0033336005,0.0025991846,0.0031241875],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002898687,0.00016703802,0.0017053548,0.0005397904,0.00005938014,0.0007245184,0.00081674336,0.028163737,0.01615151,0.76228756,0.019641563,0.169453],"study_design_scores_gemma":[0.0001588076,0.000270263,0.00045082875,0.00035051812,0.00014247211,0.0013334965,0.00014820619,0.27110088,0.057217434,0.47845593,0.19017586,0.00019531955],"about_ca_topic_score_codex":0.0037748986,"about_ca_topic_score_gemma":0.0029485673,"teacher_disagreement_score":0.011401531,"about_ca_system_score_codex":0.0013410565,"about_ca_system_score_gemma":0.00375477,"threshold_uncertainty_score":0.038141906},"labels":[],"label_agreement":null},{"id":"W1973842727","doi":"10.1145/2304510.2304517","title":"Targeted genetic test SQL generation for the DB2 database","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); Polytechnique Montréal","funders":"","keywords":"Computer science; SQL; Data mining; Generator (circuit theory); Test case; Randomness; Genetic algorithm; Database; Machine learning; Mathematics","score_opus":0.05671190167372239,"score_gpt":0.2868569559503927,"score_spread":0.23014505427667029,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1973842727","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24545242,0.00032112494,0.7396621,0.00029466156,0.000036055844,0.00038200396,0.0004082231,0.0071380157,0.0063053956],"genre_scores_gemma":[0.56267613,0.00021338572,0.4321175,0.00017462512,0.000010211192,0.00029313954,0.0009292796,0.0005036641,0.0030820887],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99923086,0.00027744152,0.000034384804,0.00011596741,0.00027688354,0.00006451604],"domain_scores_gemma":[0.9988028,0.00081352564,0.00008702618,0.0001112305,0.00015254297,0.000032890806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000987392,0.00052405236,0.000285254,0.00054509734,0.00028487513,0.0006182884,0.0008925433,0.0006166194,0.0015850714],"category_scores_gemma":[0.0026803813,0.00021736365,0.00045146767,0.000488615,0.00043640222,0.00039346208,0.00039915106,0.00054433255,0.0002875356],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005417445,0.000556127,0.008240645,0.0004035706,0.000105134306,0.0009005491,0.000500485,0.52134186,0.10140355,0.024437612,0.004739213,0.33682948],"study_design_scores_gemma":[0.00010292129,0.00054418057,0.002239754,0.000021710603,0.000033027867,0.00039368312,0.000056946672,0.9243438,0.06155236,0.003870544,0.0068084067,0.00003262948],"about_ca_topic_score_codex":0.0044018407,"about_ca_topic_score_gemma":0.0027993864,"teacher_disagreement_score":0.0044018407,"about_ca_system_score_codex":0.0006417958,"about_ca_system_score_gemma":0.0006613069,"threshold_uncertainty_score":0.008752406},"labels":[],"label_agreement":null},{"id":"W1974201381","doi":"10.1145/1463788.1463817","title":"SIFT","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); Western University","funders":"","keywords":"Computer science; TRACE (psycholinguistics); Scale-invariant feature transform; Software; Software development; Scalability; Set (abstract data type); Software system; Programming language; Artificial intelligence; Operating system; Feature extraction","score_opus":0.03447392060513783,"score_gpt":0.23421270695678428,"score_spread":0.19973878635164646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1974201381","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025252402,0.0013075435,0.74554724,0.0005869139,0.00057828106,0.00080166594,0.013959276,0.13822426,0.07374247],"genre_scores_gemma":[0.24508753,0.0011305396,0.5963466,0.0008533077,0.00020810498,0.0006646413,0.060106043,0.011060724,0.08454244],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998606,0.00008885918,0.00010177171,0.000379405,0.000648929,0.00017502977],"domain_scores_gemma":[0.998552,0.00025285475,0.0001167824,0.0005028843,0.00050053507,0.000074906784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00084737636,0.0014429786,0.0011007452,0.004218327,0.0010217932,0.002333691,0.0022009392,0.0013870291,0.06257598],"category_scores_gemma":[0.0034539783,0.00060237624,0.0012489109,0.003305431,0.00051148917,0.0033930468,0.002307977,0.00083367503,0.027868778],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006991959,0.00013189978,0.0033613937,0.0005143506,0.000121882236,0.00031044465,0.00018722976,0.0050896555,0.014979075,0.014742984,0.14239588,0.817466],"study_design_scores_gemma":[0.0003294611,0.000755512,0.010515368,0.00023287036,0.000241224,0.0031132244,0.00071722624,0.19156547,0.09553444,0.04865292,0.6480914,0.00025098858],"about_ca_topic_score_codex":0.004178274,"about_ca_topic_score_gemma":0.0055542244,"teacher_disagreement_score":0.06257598,"about_ca_system_score_codex":0.0007331434,"about_ca_system_score_gemma":0.0012406849,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W1974536975","doi":"10.1145/1639622.1639623","title":"Run-time conformance checking of mobile and distributed systems using executable models","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Executable; Computer science; Conformance testing; Unified Modeling Language; Distributed computing; Mobile device; Programming language; Operating system; Software","score_opus":0.024181503865814632,"score_gpt":0.25414855034019235,"score_spread":0.2299670464743777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1974536975","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03834092,0.0000793186,0.95729625,0.00009411792,0.00001952909,0.00012047626,0.0000529737,0.0029727204,0.0010236747],"genre_scores_gemma":[0.511695,0.00013917683,0.4856068,0.00008485006,0.000023190983,0.00022803273,0.0003387079,0.00057865883,0.0013055272],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9918528,0.0029427996,0.0006253314,0.00078104506,0.0033908873,0.00040716867],"domain_scores_gemma":[0.9843201,0.00874949,0.0016800349,0.0034484882,0.0016010501,0.0002007936],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005681603,0.0009587502,0.0007016049,0.0013137558,0.0006210956,0.0021347092,0.0022609248,0.0011708044,0.0014145627],"category_scores_gemma":[0.020956002,0.00085341884,0.0015934039,0.0005488466,0.0020110924,0.003493398,0.001849393,0.0014831835,0.0002649798],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009483456,0.00066376454,0.013633366,0.00067648187,0.00036697907,0.0025172972,0.0030910678,0.5766005,0.062457144,0.20766146,0.0015968703,0.12978676],"study_design_scores_gemma":[0.00010219848,0.00030220777,0.0006275838,0.000122581,0.00009775899,0.00044005289,0.0001789141,0.90965825,0.03975475,0.042300835,0.00636678,0.0000480515],"about_ca_topic_score_codex":0.0033748446,"about_ca_topic_score_gemma":0.0038985743,"teacher_disagreement_score":0.005681603,"about_ca_system_score_codex":0.0011931037,"about_ca_system_score_gemma":0.0018952173,"threshold_uncertainty_score":0.030047536},"labels":[],"label_agreement":null},{"id":"W1975057653","doi":"10.1142/s0218194010004621","title":"ASSESSING TEST SUITES FOR BUFFER OVERFLOW VULNERABILITIES","year":2010,"lang":"en","type":"article","venue":"International Journal of Software Engineering and Knowledge Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Buffer overflow; Test suite; Computer science; Secure coding; Software security assurance; Benchmark (surveying); Test case; Mutation; Software; Exploit; Reliability engineering; Vulnerability (computing); Computer security; Engineering; Operating system; Machine learning; Information security","score_opus":0.01213610208872551,"score_gpt":0.27433133318351743,"score_spread":0.2621952310947919,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1975057653","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8937856,0.0003586048,0.102953084,0.00011566354,0.000017528757,0.00016893394,0.00026897914,0.001375577,0.00095594855],"genre_scores_gemma":[0.9328277,0.00016039182,0.0657686,0.000030362613,0.000010136118,0.00015222133,0.0007835312,0.000089556816,0.00017753012],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99580103,0.0018424576,0.00037985324,0.00033278627,0.0014084002,0.00023545363],"domain_scores_gemma":[0.9707907,0.02297006,0.0024742023,0.0011186627,0.0021659592,0.00048040846],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003912335,0.001228429,0.00055730686,0.004355071,0.00024777776,0.00066655583,0.00087294553,0.0006830472,0.00061820203],"category_scores_gemma":[0.029861672,0.000252663,0.000752594,0.0014601084,0.0005428386,0.00078580563,0.00067818497,0.00041000926,0.000095931566],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009308841,0.0010901649,0.12298425,0.0006676735,0.0004951348,0.0016726378,0.0008398492,0.48439234,0.0918817,0.0071611106,0.0015119296,0.28637227],"study_design_scores_gemma":[0.00013643366,0.0019145401,0.02633415,0.00007475734,0.0002566339,0.0009271484,0.0002634693,0.9141574,0.04862406,0.0058575594,0.0013970354,0.000056880795],"about_ca_topic_score_codex":0.0013044905,"about_ca_topic_score_gemma":0.0012296851,"teacher_disagreement_score":0.004355071,"about_ca_system_score_codex":0.0005967729,"about_ca_system_score_gemma":0.0007035385,"threshold_uncertainty_score":0.02069062},"labels":[],"label_agreement":null},{"id":"W1975319166","doi":"10.1145/1321211.1321239","title":"A test framework for integration testing of object-oriented programs","year":2007,"lang":"en","type":"article","venue":"Proceedings of CASCON","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Integration testing; Unified Modeling Language; Sequence diagram; Software engineering; Class diagram; Implementation; Programming language; White-box testing; Java; Object-oriented programming; Test case; Automation; Test Management Approach; Test strategy; Model-based testing; Software development; Software; Engineering; Machine learning","score_opus":0.034891224725824996,"score_gpt":0.2971395826582922,"score_spread":0.2622483579324672,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1975319166","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016507162,0.00017260737,0.9931937,0.00010846944,0.00003046951,0.00014324303,0.000045930206,0.0035820254,0.0010728926],"genre_scores_gemma":[0.071720034,0.00034934696,0.92456144,0.00014140156,0.00006662192,0.00057946885,0.00043928303,0.00079748884,0.0013448304],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99239093,0.0026769661,0.00087940105,0.0007995905,0.0027887935,0.00046436055],"domain_scores_gemma":[0.99146616,0.0048796176,0.00062415155,0.0012841657,0.0013027374,0.00044315928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008123491,0.0017902853,0.001544391,0.0039150952,0.0010287028,0.0024847018,0.0031307838,0.0019876182,0.0028716035],"category_scores_gemma":[0.013941519,0.0009743459,0.002291812,0.001828941,0.0026839555,0.002814507,0.002008298,0.0028940442,0.00094300497],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029821022,0.00047385818,0.0031538124,0.0008909575,0.00021267503,0.0020429299,0.0008788648,0.10702479,0.023554394,0.53169304,0.011050633,0.31872585],"study_design_scores_gemma":[0.00025799958,0.0006085977,0.0016446547,0.0007901998,0.00020605969,0.0026447806,0.0002015835,0.62817764,0.025608245,0.22585402,0.113803245,0.00020295075],"about_ca_topic_score_codex":0.0047650198,"about_ca_topic_score_gemma":0.0027137322,"teacher_disagreement_score":0.008123491,"about_ca_system_score_codex":0.001355554,"about_ca_system_score_gemma":0.0025829428,"threshold_uncertainty_score":0.042961597},"labels":[],"label_agreement":null},{"id":"W1975581743","doi":"10.1109/icst.2012.107","title":"Generating String Test Data for Code Coverage","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada; University of Nebraska-Lincoln","keywords":"Computer science; Programming language; Grammar; Code coverage; String (physics); Heuristic; Code (set theory); Rule-based machine translation; Code generation; Java; Theoretical computer science; Natural language processing; Artificial intelligence; Mathematics; Set (abstract data type); Software; Linguistics","score_opus":0.10267725710357246,"score_gpt":0.3291761569067954,"score_spread":0.22649889980322296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1975581743","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23101804,0.00020530073,0.74614215,0.00059201964,0.000070990376,0.0004218781,0.0036358805,0.01163833,0.0062754564],"genre_scores_gemma":[0.67778665,0.00011318985,0.3123898,0.00026759607,0.00004565492,0.0007553867,0.0062586567,0.0015336459,0.0008494512],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9889798,0.0042077336,0.00093057536,0.0011995385,0.0042526233,0.00042977653],"domain_scores_gemma":[0.92217416,0.059939586,0.0030253404,0.008507451,0.005920048,0.0004334586],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053755403,0.0009149431,0.001072482,0.005282251,0.0005164711,0.0015752015,0.0016053229,0.0015881124,0.0034227544],"category_scores_gemma":[0.06521987,0.00042373044,0.001096798,0.004044981,0.001401226,0.0027645049,0.0021383017,0.0012561836,0.0007505496],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013451392,0.00075929624,0.05510363,0.0008260934,0.00025179327,0.0014473924,0.00091140415,0.24870867,0.052665323,0.05650613,0.013797944,0.56767726],"study_design_scores_gemma":[0.00019379867,0.0004611714,0.0076287915,0.00011691976,0.00007444391,0.00068832596,0.00020626967,0.8494146,0.07610527,0.058007877,0.0070233117,0.00007922836],"about_ca_topic_score_codex":0.0013070848,"about_ca_topic_score_gemma":0.0010809978,"teacher_disagreement_score":0.0053755403,"about_ca_system_score_codex":0.0009214529,"about_ca_system_score_gemma":0.0013867907,"threshold_uncertainty_score":0.028428912},"labels":[],"label_agreement":null},{"id":"W1977244105","doi":"10.1145/2559936","title":"Automated cookie collection testing","year":2014,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; The King's University","funders":"","keywords":"Computer science; Web testing; Web application; Web application security; Security testing; Software engineering; World Wide Web; Web development; Web service; Operating system; Cloud computing","score_opus":0.08726401518129814,"score_gpt":0.3104312761742673,"score_spread":0.22316726099296916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1977244105","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4839038,0.0009765662,0.46978024,0.00035099464,0.00014084086,0.00090808864,0.0012172062,0.026184622,0.016537528],"genre_scores_gemma":[0.8801357,0.00022849235,0.11233967,0.00018106736,0.000026391554,0.00031217726,0.0013023212,0.00058814837,0.0048859594],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9953499,0.00096770644,0.00021365225,0.0006838789,0.0024059846,0.00037883534],"domain_scores_gemma":[0.9845718,0.008094545,0.0013963344,0.0029258835,0.002731422,0.000279973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013070145,0.0010756274,0.0006867674,0.0020648637,0.0004979591,0.0010974501,0.0014814029,0.0007557955,0.004114399],"category_scores_gemma":[0.010545782,0.00038622908,0.0004897864,0.0011142652,0.00079659215,0.0013664634,0.0012050867,0.00070267613,0.0011269667],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094921776,0.00094733125,0.024419079,0.0005958052,0.00012352901,0.0010809035,0.00080059026,0.056714185,0.16883667,0.008604217,0.010272431,0.726656],"study_design_scores_gemma":[0.00023110172,0.0022208903,0.035064336,0.00022423961,0.00015008533,0.0027224845,0.0004994121,0.6159855,0.28911078,0.021313023,0.032221515,0.00025659302],"about_ca_topic_score_codex":0.0043498846,"about_ca_topic_score_gemma":0.0046205134,"teacher_disagreement_score":0.0043498846,"about_ca_system_score_codex":0.0007255694,"about_ca_system_score_gemma":0.0017783832,"threshold_uncertainty_score":0.013764024},"labels":[],"label_agreement":null},{"id":"W1978219123","doi":"10.5555/2818754.2818796","title":"Detecting inconsistencies in JavaScript MVC applications","year":2015,"lang":"en","type":"article","venue":"International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; JavaScript; Web application; Consistency (knowledge bases); Model–view–controller; Identifier; Data mining; Software engineering; Programming language; Operating system; Artificial intelligence; User interface","score_opus":0.07089629372013515,"score_gpt":0.2861331090805849,"score_spread":0.21523681536044978,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978219123","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55587286,0.0013182038,0.38000095,0.00035095302,0.00015139989,0.00034930903,0.0012089285,0.058221076,0.0025263028],"genre_scores_gemma":[0.7567532,0.00022710636,0.23873039,0.00018285685,0.00003955296,0.00009087413,0.0020556669,0.0010417479,0.00087854953],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9893679,0.0017141177,0.0010588187,0.0018387274,0.0056290147,0.00039152912],"domain_scores_gemma":[0.9601535,0.01822336,0.0070634037,0.005150009,0.008856346,0.00055333105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004541399,0.0010603347,0.0011816246,0.0052635707,0.0006098601,0.0017261445,0.0020983666,0.0015743792,0.00057577755],"category_scores_gemma":[0.0312829,0.00069848093,0.0005123048,0.002410567,0.0005556383,0.0016803538,0.0017779294,0.0009931594,0.00045024586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011756151,0.0008533814,0.221593,0.0012748858,0.00036945444,0.00271519,0.0036146215,0.032243818,0.10638913,0.004511956,0.012377912,0.61288095],"study_design_scores_gemma":[0.00011550947,0.00057554326,0.07990353,0.0002714601,0.0002338205,0.0031794587,0.0008433025,0.74397326,0.14952095,0.0050280676,0.016143808,0.00021124043],"about_ca_topic_score_codex":0.0041686147,"about_ca_topic_score_gemma":0.0043159337,"teacher_disagreement_score":0.0052635707,"about_ca_system_score_codex":0.00078160025,"about_ca_system_score_gemma":0.0010952131,"threshold_uncertainty_score":0.024017513},"labels":[],"label_agreement":null},{"id":"W1979107890","doi":"10.1109/compsac.2012.50","title":"On Capturing Effects of Modifications as Data Dependencies","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Test suite; Computer science; Extended finite-state machine; Suite; Regression testing; Data mining; Representation (politics); Set (abstract data type); Test case; Finite-state machine; Theoretical computer science; Regression analysis; Algorithm; Machine learning; Software; Programming language; Software system","score_opus":0.055943871091449014,"score_gpt":0.3048740989086774,"score_spread":0.24893022781722837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1979107890","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02776236,0.00030283036,0.9663472,0.0004239262,0.000037009595,0.00017271942,0.00022838416,0.0008798719,0.0038456614],"genre_scores_gemma":[0.5020791,0.0011327347,0.4912727,0.0005732805,0.0001424449,0.00060862646,0.000669857,0.00068189,0.0028393613],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.994588,0.002016588,0.00036570404,0.00061693584,0.0020778798,0.00033498235],"domain_scores_gemma":[0.94398355,0.039939035,0.0030376813,0.009308696,0.003392353,0.00033863005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046395925,0.0017717936,0.0009668067,0.0025654717,0.00078663573,0.0019979503,0.0016293692,0.0014515871,0.0024802703],"category_scores_gemma":[0.039154388,0.0009879373,0.001217578,0.0024080777,0.004571965,0.009890902,0.0024182987,0.003471296,0.0003570646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040516906,0.00023399513,0.00879625,0.000622043,0.00010166966,0.0010388419,0.0017463586,0.29243314,0.024328636,0.502213,0.0018915957,0.16618934],"study_design_scores_gemma":[0.000043364893,0.00020704676,0.002848046,0.0002258344,0.00014008174,0.0005741821,0.00014891104,0.61375815,0.019032976,0.35394412,0.008979604,0.00009763385],"about_ca_topic_score_codex":0.0039425762,"about_ca_topic_score_gemma":0.004146049,"teacher_disagreement_score":0.0046395925,"about_ca_system_score_codex":0.0014070442,"about_ca_system_score_gemma":0.0018715122,"threshold_uncertainty_score":0.024536848},"labels":[],"label_agreement":null},{"id":"W1980713699","doi":"10.1145/2610384.2610406","title":"DOM-based test adequacy criteria for web applications","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Computer science; Code coverage; Granularity; Code (set theory); Set (abstract data type); Web application; Data mining; Test (biology); Quality (philosophy); Web testing; Measure (data warehouse); Information retrieval; Database; Web page; World Wide Web; Web application security; Software; Web development; Programming language","score_opus":0.02655093496977057,"score_gpt":0.30859088909850024,"score_spread":0.2820399541287297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1980713699","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21290225,0.00069059164,0.7728797,0.0003274609,0.00004138295,0.0005237053,0.0011599261,0.0057566706,0.0057182675],"genre_scores_gemma":[0.80182046,0.000095975025,0.19542828,0.000106594125,0.000035900888,0.00047912655,0.0011722173,0.00037590365,0.0004854271],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9777348,0.007961825,0.0027124146,0.0012732579,0.009505131,0.0008124755],"domain_scores_gemma":[0.87877446,0.08486255,0.009682507,0.006322072,0.018544039,0.0018143322],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01000015,0.0009831833,0.0008852438,0.009746417,0.0007250524,0.002392482,0.0012366499,0.0013594552,0.0017691044],"category_scores_gemma":[0.12132722,0.00040270516,0.0006777388,0.0022646463,0.0013895539,0.002803625,0.0024691008,0.0009110729,0.00048683118],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001989133,0.00089274073,0.18763447,0.0015187002,0.00035768354,0.0018553458,0.0024541593,0.22826962,0.08174867,0.054023784,0.010032502,0.42922318],"study_design_scores_gemma":[0.00009580066,0.00046931716,0.035725195,0.0002744778,0.00006491451,0.0010927322,0.00044354904,0.8996706,0.032577023,0.023372738,0.006101727,0.00011191903],"about_ca_topic_score_codex":0.002217796,"about_ca_topic_score_gemma":0.0023913714,"teacher_disagreement_score":0.01000015,"about_ca_system_score_codex":0.00097555586,"about_ca_system_score_gemma":0.0009078316,"threshold_uncertainty_score":0.052886486},"labels":[],"label_agreement":null},{"id":"W1982395839","doi":"10.1002/1099-1689(200009)10:3<149::aid-stvr206>3.0.co;2-t","title":"State generation and automated class testing","year":2000,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Container (type theory); Computer science; White-box testing; Tree (set theory); Class (philosophy); Block (permutation group theory); Java; Reliability (semiconductor); Black box; Code coverage; Reliability engineering; Data mining; Programming language; Engineering; Artificial intelligence; Software; Software development","score_opus":0.03457682983096203,"score_gpt":0.2649357170513301,"score_spread":0.2303588872203681,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982395839","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02540016,0.000054070206,0.96835977,0.00008360656,0.000015593425,0.00010705565,0.00006170208,0.0037406415,0.002177354],"genre_scores_gemma":[0.360162,0.00012209876,0.6353642,0.00010598236,0.000019416097,0.000350787,0.00057307124,0.0007585954,0.0025439286],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99756515,0.0009266269,0.00012594827,0.00039912565,0.00081143965,0.00017178302],"domain_scores_gemma":[0.9918333,0.0054689124,0.0007414915,0.0012179198,0.0006404644,0.00009786544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017572932,0.0008777852,0.00069743465,0.0013587369,0.00046451512,0.0012780859,0.0015793766,0.0008086043,0.004091278],"category_scores_gemma":[0.010765,0.00043681465,0.00072004955,0.00090052735,0.0011911046,0.0018529606,0.0011173487,0.0007587568,0.0010063564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003560525,0.00025513454,0.0038520365,0.00024637248,0.000053177824,0.00035757152,0.0003510965,0.22275989,0.02737512,0.08643515,0.004839147,0.65311927],"study_design_scores_gemma":[0.00007693913,0.00012607595,0.00064371957,0.00003073677,0.000021718159,0.00023840189,0.000035382865,0.9267737,0.029762425,0.037156396,0.005107242,0.000027211905],"about_ca_topic_score_codex":0.0016801618,"about_ca_topic_score_gemma":0.0015235969,"teacher_disagreement_score":0.004091278,"about_ca_system_score_codex":0.0010525269,"about_ca_system_score_gemma":0.0011519399,"threshold_uncertainty_score":0.013686717},"labels":[],"label_agreement":null},{"id":"W1984849298","doi":"10.1109/cseet.2010.29","title":"Current State of the Software Testing Education in North American Academia and Some Recommendations for the New Educators","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; Florida Institute of Technology; University of Florida; Purdue University","keywords":"Software testing; Curriculum; Software peer review; Software; Software engineering; State (computer science); System integration testing; Computer science; Engineering management; Engineering ethics; Software development; Medical education; Software construction; Engineering; Psychology; Pedagogy; Medicine","score_opus":0.028580521714709435,"score_gpt":0.326672967562545,"score_spread":0.29809244584783556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1984849298","genre_codex":"commentary","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03888225,0.2054665,0.010853116,0.67637974,0.0041562566,0.00015174484,0.00044066855,0.0013794282,0.06229035],"genre_scores_gemma":[0.49159807,0.3432108,0.050558206,0.06470682,0.0023879854,0.00029842116,0.001245369,0.00038096422,0.04561335],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98606247,0.0028956607,0.0011230956,0.0011618919,0.006866911,0.0018899653],"domain_scores_gemma":[0.90939736,0.021538153,0.0058288183,0.0017363318,0.045029517,0.016469875],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021021826,0.0004915002,0.00060524966,0.0059885443,0.0063581965,0.00792195,0.003090506,0.0026379968,0.010669525],"category_scores_gemma":[0.025911078,0.0004159172,0.00043126856,0.007158168,0.006654681,0.005962238,0.0030466933,0.00346266,0.0018495799],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009881259,0.00032207783,0.019262388,0.002008108,0.000013374782,0.00019242677,0.0023979382,0.00064355147,0.0008338113,0.024390228,0.085955665,0.86388165],"study_design_scores_gemma":[0.00005202435,0.0004376165,0.09534532,0.012161748,0.000119856835,0.0010059,0.035820473,0.0022526246,0.0022722578,0.021962626,0.82828116,0.00028830193],"about_ca_topic_score_codex":0.33624703,"about_ca_topic_score_gemma":0.47672448,"teacher_disagreement_score":0.33624703,"about_ca_system_score_codex":0.024501812,"about_ca_system_score_gemma":0.07442615,"threshold_uncertainty_score":0.6685797},"labels":[],"label_agreement":null},{"id":"W1986024814","doi":"10.1007/s11219-013-9203-5","title":"BlackHorse: creating smart test cases from brittle recorded tests","year":2013,"lang":"en","type":"article","venue":"Software Quality Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Blackberry (Canada); Western University","funders":"","keywords":"Computer science; Keyword-driven testing; Java; Test case; Test Management Approach; Manual testing; Code coverage; Regression testing; Test (biology); Software engineering; Graphical user interface testing; Reliability engineering; Test harness; Programming language; Software; Engineering; Software development; User interface; Software construction; Machine learning","score_opus":0.044124431323995406,"score_gpt":0.31495280582406704,"score_spread":0.27082837450007163,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1986024814","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033777513,0.00021508275,0.82454026,0.00018116015,0.00011557808,0.0004656759,0.0010966307,0.13537996,0.004228118],"genre_scores_gemma":[0.2942387,0.00024183569,0.6738008,0.00021027063,0.000045758028,0.0005203633,0.0046327976,0.020613696,0.005695749],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977412,0.0005917949,0.00015543713,0.00036742017,0.00096922595,0.00017488837],"domain_scores_gemma":[0.98835886,0.007225937,0.00072148436,0.0022320584,0.0011453927,0.0003162838],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023490363,0.0024561945,0.0009229177,0.0027395228,0.00049844844,0.0019686327,0.0033642473,0.0015302007,0.013573176],"category_scores_gemma":[0.018810036,0.001498685,0.0014864603,0.0010803312,0.0011842557,0.0029031676,0.00281084,0.0016184564,0.0032854243],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014502716,0.0006839823,0.015740883,0.0015009085,0.00043128702,0.0048092166,0.0023514875,0.10821217,0.060018845,0.018552974,0.0529854,0.7332625],"study_design_scores_gemma":[0.00037221352,0.000397229,0.0033766734,0.0002009579,0.00016384978,0.0016141195,0.00034071432,0.86888295,0.07875921,0.021585992,0.02416231,0.00014371693],"about_ca_topic_score_codex":0.0022102871,"about_ca_topic_score_gemma":0.0038655482,"teacher_disagreement_score":0.013573176,"about_ca_system_score_codex":0.0005201297,"about_ca_system_score_gemma":0.00093828625,"threshold_uncertainty_score":0.04540676},"labels":[],"label_agreement":null},{"id":"W1986867802","doi":"10.1016/j.entcs.2007.08.002","title":"Can a Model Checker Generate Tests for Non-Deterministic Systems?","year":2007,"lang":"en","type":"article","venue":"Electronic Notes in Theoretical Computer Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Model checking; Computer science; Determinism; Modular design; Counterexample; Model-based testing; Programming language; Theoretical computer science; CTL*; Temporal logic; Software; Distributed computing; Test case; Mathematics; Machine learning","score_opus":0.017236272022157544,"score_gpt":0.2968998082688012,"score_spread":0.27966353624664364,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1986867802","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014331633,0.00023401718,0.97865796,0.0019859897,0.0001586107,0.000042940435,0.00004104331,0.0027435059,0.0018042825],"genre_scores_gemma":[0.5147948,0.00051320676,0.48018473,0.0010492427,0.00016468529,0.00017753855,0.00025937226,0.0007489583,0.002107416],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99629885,0.0016591738,0.00015606082,0.00043809996,0.0011367529,0.00031106925],"domain_scores_gemma":[0.9724708,0.021291547,0.001004658,0.0035554117,0.0014366915,0.00024096211],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039294004,0.00086801325,0.0008033151,0.001020972,0.00037377444,0.0014221782,0.0016673426,0.002395656,0.0034695792],"category_scores_gemma":[0.047746014,0.00053440395,0.0011409036,0.00070470007,0.0020575204,0.005342805,0.0011863444,0.0019552743,0.001004939],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066691206,0.0006241661,0.008650082,0.00079857034,0.00025014917,0.00090864394,0.00050446106,0.22597408,0.027557239,0.34853503,0.012213318,0.37331727],"study_design_scores_gemma":[0.00016851704,0.00017854232,0.00043778997,0.00013205815,0.00006492965,0.00035694113,0.00007179343,0.7093213,0.021043077,0.2623596,0.005817959,0.000047427497],"about_ca_topic_score_codex":0.0011169539,"about_ca_topic_score_gemma":0.0014642564,"teacher_disagreement_score":0.0039294004,"about_ca_system_score_codex":0.0006120796,"about_ca_system_score_gemma":0.0011191624,"threshold_uncertainty_score":0.020780921},"labels":[],"label_agreement":null},{"id":"W1986927667","doi":"10.1007/s11265-013-0845-0","title":"Enhancing Hardware Assisted Test Insertion Capabilities on Embedded Processors using an FPGA-based Agile Test Support Co-processor","year":2013,"lang":"en","type":"article","venue":"Journal of Signal Processing Systems","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Alberta; University of Calgary","funders":"","keywords":"Agile software development; Embedded system; Debugging; Computer science; Overhead (engineering); Field-programmable gate array; Software; Test (biology); Computer hardware; Operating system; Computer architecture; Software engineering","score_opus":0.04360086151967359,"score_gpt":0.2980660064033549,"score_spread":0.2544651448836813,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1986927667","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5704319,0.0003062163,0.42040434,0.00013680267,0.000083673905,0.00009758802,0.00004951207,0.0028754403,0.005614644],"genre_scores_gemma":[0.937425,0.000047044927,0.06098,0.000059878348,0.0000125005345,0.000019368039,0.000030442465,0.000056854762,0.0013688741],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997733,0.000041764357,0.000013112318,0.000035514382,0.000087775545,0.000048586357],"domain_scores_gemma":[0.9991579,0.0002711427,0.00011221422,0.0001434574,0.00025540384,0.000059931484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00019985944,0.00043294835,0.00020196223,0.00045471324,0.00018232349,0.000385378,0.00068264926,0.000296116,0.0018804498],"category_scores_gemma":[0.0007097766,0.00014642933,0.00011984996,0.00023068217,0.00014670716,0.00045577902,0.00036911693,0.00042201235,0.00041125374],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013880259,0.0002933159,0.0057510757,0.00014536823,0.00004871033,0.0007290787,0.00018891421,0.018351378,0.69117224,0.002627932,0.0013224126,0.2779816],"study_design_scores_gemma":[0.00013466191,0.002519686,0.0053350786,0.000039567272,0.00009486158,0.0023301921,0.00007215655,0.2744202,0.7071911,0.0008030641,0.007018887,0.000040605293],"about_ca_topic_score_codex":0.00031447574,"about_ca_topic_score_gemma":0.00076761417,"teacher_disagreement_score":0.0018804498,"about_ca_system_score_codex":0.0001364885,"about_ca_system_score_gemma":0.00029899427,"threshold_uncertainty_score":0.006290674},"labels":[],"label_agreement":null},{"id":"W1987534773","doi":"10.1007/s10515-010-0079-3","title":"Example-based model-transformation testing","year":2011,"lang":"en","type":"article","venue":"Automated Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Université de Montréal","funders":"","keywords":"Transformation (genetics); Oracle; Computer science; Model transformation; Function (biology); Premise; Quality (philosophy); Data mining; Reliability engineering; Artificial intelligence; Theoretical computer science; Industrial engineering; Machine learning; Software engineering; Engineering","score_opus":0.06099352935504673,"score_gpt":0.23204957400922685,"score_spread":0.17105604465418012,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1987534773","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06666968,0.00008717061,0.9161105,0.00031826305,0.000059259983,0.00013287114,0.0001964922,0.00878639,0.0076393364],"genre_scores_gemma":[0.60356766,0.00008758546,0.39169598,0.0001528593,0.00001493752,0.00010913188,0.0004578217,0.00088556594,0.0030284692],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963246,0.0015672987,0.00015918819,0.0003253329,0.00139993,0.00022357519],"domain_scores_gemma":[0.9882947,0.007674256,0.00030508635,0.0025023343,0.001086942,0.0001365485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016400971,0.0009904329,0.0006688369,0.0009195866,0.00042158572,0.00087332434,0.0025014873,0.0015312997,0.008687332],"category_scores_gemma":[0.015656622,0.00048183795,0.0009460033,0.00079694745,0.0009871728,0.0026224484,0.0017246233,0.0014098133,0.0010114103],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016733431,0.0013299254,0.0067462455,0.000813531,0.00025011398,0.001549371,0.0006182463,0.32726723,0.05380504,0.09504671,0.012620493,0.49827978],"study_design_scores_gemma":[0.00015513449,0.00019415664,0.00051536184,0.000038520324,0.000059801117,0.00043118157,0.0000526757,0.93235075,0.028553257,0.034933295,0.0026912559,0.000024519077],"about_ca_topic_score_codex":0.001730677,"about_ca_topic_score_gemma":0.0029588211,"teacher_disagreement_score":0.008687332,"about_ca_system_score_codex":0.0005205582,"about_ca_system_score_gemma":0.0007761983,"threshold_uncertainty_score":0.029062092},"labels":[],"label_agreement":null},{"id":"W1987538235","doi":"10.1145/1297846.1297882","title":"Open framework for conformance testing via scenarios","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Conformance testing; Computer science; Software engineering; Software testing; Formal specification; Conformance checking; Software; Reliability engineering; Programming language; Operating system; Engineering; Compatibility (geochemistry); Business process; Standardization","score_opus":0.07472432542418293,"score_gpt":0.35178939586205554,"score_spread":0.2770650704378726,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1987538235","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00089529296,0.00011252672,0.9924334,0.0003552512,0.00006429532,0.00025529938,0.00007618747,0.0026219236,0.0031858347],"genre_scores_gemma":[0.05161491,0.0004298883,0.94153297,0.00038262893,0.0001791373,0.0012880935,0.00064423995,0.0010205685,0.0029075872],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9689831,0.012058653,0.0035410156,0.002869626,0.010949111,0.001598408],"domain_scores_gemma":[0.97237355,0.013163923,0.001441812,0.0073374202,0.004159351,0.0015240542],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02568016,0.002076293,0.0018608076,0.004245043,0.0022755556,0.007526408,0.00775917,0.0053054946,0.008597873],"category_scores_gemma":[0.031933714,0.0018889506,0.0042695072,0.0019893441,0.0068240487,0.013395748,0.0081309555,0.009150203,0.0022388434],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008694839,0.00019433521,0.00045702522,0.00020321805,0.00006923073,0.00046042464,0.0005121162,0.013494865,0.002123922,0.9221927,0.0038547737,0.05635041],"study_design_scores_gemma":[0.00015273792,0.00014467158,0.0002533369,0.0004054123,0.000075186224,0.00063234975,0.00017371602,0.12117149,0.0057348274,0.76067,0.11044217,0.00014405164],"about_ca_topic_score_codex":0.0027347421,"about_ca_topic_score_gemma":0.0021814923,"teacher_disagreement_score":0.02568016,"about_ca_system_score_codex":0.0021517577,"about_ca_system_score_gemma":0.004129567,"threshold_uncertainty_score":0.13581127},"labels":[],"label_agreement":null},{"id":"W1988040396","doi":"10.1109/icst.2012.120","title":"Automated Unit Testing of a SCADA Control Software: An Industrial Case Study Based on Action Research","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"SCADA; Test suite; Unit testing; Regression testing; White-box testing; Computer science; Keyword-driven testing; Test Management Approach; Reliability engineering; Test (biology); Suite; Manual testing; Software engineering; Rocket (weapon); Software; Black box; Engineering; Test case; Software development; Software construction; Operating system; Regression analysis; Artificial intelligence; Machine learning; Aeronautics","score_opus":0.3842958960814827,"score_gpt":0.4381098421641235,"score_spread":0.05381394608264084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1988040396","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9278869,0.00022474046,0.063420564,0.00050313125,0.000018490813,0.00052437076,0.00012034227,0.00047784793,0.00682354],"genre_scores_gemma":[0.9474904,0.00013978206,0.05008533,0.00008725734,0.0000112157,0.0001973459,0.00010775018,0.000058744383,0.0018222268],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9930976,0.0044924063,0.000256732,0.0005349752,0.0012149565,0.00040329993],"domain_scores_gemma":[0.97304386,0.020897493,0.0013440745,0.0020658255,0.0017341111,0.00091468333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004577037,0.0008580875,0.00048454554,0.0013811745,0.0016059523,0.001203835,0.0020132018,0.0022445815,0.0017738298],"category_scores_gemma":[0.013704881,0.00036974892,0.00054797577,0.0008542382,0.0021681932,0.0011676685,0.0012452246,0.0009946437,0.00037839435],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025972642,0.01765485,0.13973945,0.0023526114,0.0003904889,0.07343093,0.081817895,0.13720426,0.06920857,0.03947002,0.0081730215,0.42796078],"study_design_scores_gemma":[0.0021886653,0.023044616,0.08295395,0.001123715,0.0006029823,0.043887485,0.052932505,0.52718604,0.15721351,0.01756889,0.09073964,0.00055803213],"about_ca_topic_score_codex":0.0044190725,"about_ca_topic_score_gemma":0.0061087357,"teacher_disagreement_score":0.004577037,"about_ca_system_score_codex":0.0012906995,"about_ca_system_score_gemma":0.0010809411,"threshold_uncertainty_score":0.024205983},"labels":[],"label_agreement":null},{"id":"W1989734351","doi":"10.1145/2614106.2614206","title":"OpenVL","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science","score_opus":0.010473633179819484,"score_gpt":0.22952648707943524,"score_spread":0.21905285389961576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1989734351","genre_codex":"other","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020016038,0.0072526312,0.01969727,0.0076395017,0.0069723767,0.00046558015,0.037263006,0.01660155,0.9021064],"genre_scores_gemma":[0.012895445,0.0037704476,0.005016511,0.0014200555,0.00077044714,0.000340474,0.039495554,0.002967964,0.93332297],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99845433,0.00023575543,0.00010229315,0.00031346452,0.0006808427,0.00021335323],"domain_scores_gemma":[0.9963278,0.0006849691,0.00015713646,0.00086789473,0.0011241196,0.0008379617],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018724157,0.0012341952,0.001785666,0.0024850387,0.0009856434,0.0044622943,0.0039472943,0.0035564627,0.81471384],"category_scores_gemma":[0.0068926765,0.0007254885,0.0008498516,0.002019428,0.0009401046,0.003191917,0.00404449,0.0025141037,0.71391606],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057857146,0.00022682002,0.0003382454,0.0011848176,0.000032274973,0.00023477485,0.000084528045,0.00020656502,0.00292424,0.010684319,0.62954223,0.3539627],"study_design_scores_gemma":[0.00005174724,0.00005836137,0.00017806113,0.00017397119,0.000012570442,0.0001095279,0.000029355371,0.00011435017,0.00080990273,0.0019939875,0.9964588,0.000009269782],"about_ca_topic_score_codex":0.0022105416,"about_ca_topic_score_gemma":0.0019382631,"teacher_disagreement_score":0.81471384,"about_ca_system_score_codex":0.0012491435,"about_ca_system_score_gemma":0.002176841,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W1991113092","doi":"10.1109/issre.2014.15","title":"An Orchestrated Survey of Available Algorithms and Tools for Combinatorial Testing","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Algorithm; Test suite; Computer science; Suite; Software; Process (computing); Domain (mathematical analysis); Selection (genetic algorithm); Random testing; Test case; Data mining; Machine learning; Mathematics; Programming language","score_opus":0.13194195746342552,"score_gpt":0.3054927914941669,"score_spread":0.1735508340307414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1991113092","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0058761677,0.083867475,0.8778554,0.0009826727,0.0002657908,0.00039405716,0.00072390086,0.004357356,0.025677301],"genre_scores_gemma":[0.039653957,0.06594348,0.8852346,0.00056643574,0.00032884098,0.0007194266,0.002596939,0.0015092015,0.0034470719],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9916087,0.0022434164,0.0011407211,0.0014709239,0.0031485825,0.0003875908],"domain_scores_gemma":[0.97665524,0.01749573,0.00094861956,0.0028887442,0.001783769,0.00022783224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049161036,0.0031342753,0.002120958,0.009738463,0.0008089758,0.0044402783,0.004532425,0.0019935453,0.009507009],"category_scores_gemma":[0.021462189,0.0017000044,0.0028067424,0.013542835,0.0016304922,0.0057809697,0.0018985819,0.002647476,0.0049246224],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011167722,0.0001850915,0.0013226335,0.0052906782,0.000102239464,0.00017126623,0.00016553247,0.018100353,0.0030872454,0.066054374,0.007707772,0.8977011],"study_design_scores_gemma":[0.00020565957,0.00076699205,0.004104863,0.0076583284,0.00044747785,0.0050020125,0.00048866676,0.1612488,0.024046645,0.25789577,0.537764,0.00037070306],"about_ca_topic_score_codex":0.0014472986,"about_ca_topic_score_gemma":0.0011976875,"teacher_disagreement_score":0.009738463,"about_ca_system_score_codex":0.0016898111,"about_ca_system_score_gemma":0.0026182611,"threshold_uncertainty_score":0.031804085},"labels":[],"label_agreement":null},{"id":"W1992815253","doi":"10.1007/s10515-009-0061-0","title":"Generating a checking sequence with a minimum number of reset transitions","year":2009,"lang":"en","type":"article","venue":"Automated Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Engineering and Physical Sciences Research Council; Natural Sciences and Engineering Research Council of Canada; Leverhulme Trust","keywords":"Reset (finance); Sequence (biology); Model checking; State (computer science); Computer science; Finite-state machine; Algorithm","score_opus":0.015982541555149138,"score_gpt":0.2662486012595593,"score_spread":0.2502660597044102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1992815253","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1512382,0.00011907376,0.83154655,0.0002594973,0.00010279288,0.00045787558,0.0004417398,0.013830856,0.0020035128],"genre_scores_gemma":[0.53853035,0.00005051324,0.45753285,0.00015146886,0.000028137116,0.000252102,0.00085644866,0.001233433,0.0013646786],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9972868,0.00075533753,0.00024628246,0.0007374191,0.0006909432,0.00028316243],"domain_scores_gemma":[0.97555465,0.016043134,0.0014269459,0.003927825,0.002499143,0.0005482321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015276225,0.0014001749,0.001233728,0.00206321,0.0007005839,0.0008470123,0.0016618463,0.0018537659,0.0061159483],"category_scores_gemma":[0.01630315,0.0010541722,0.0012353632,0.00093065255,0.0008335245,0.0015760618,0.001265933,0.0015063982,0.0013876111],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0042134435,0.0010138742,0.010030819,0.00155705,0.00025195515,0.002250025,0.0006024173,0.14545523,0.256837,0.014999165,0.00673533,0.5560537],"study_design_scores_gemma":[0.0006147488,0.00096307485,0.0031566534,0.00014208918,0.0002600474,0.0012020305,0.00013031514,0.73922586,0.2238944,0.026380783,0.0039290963,0.00010086062],"about_ca_topic_score_codex":0.00092573237,"about_ca_topic_score_gemma":0.002079468,"teacher_disagreement_score":0.0061159483,"about_ca_system_score_codex":0.00043989994,"about_ca_system_score_gemma":0.0031335545,"threshold_uncertainty_score":0.02045989},"labels":[],"label_agreement":null},{"id":"W1993945834","doi":"10.1049/ic.2013.0038","title":"Prototype test insertion co-processor for agile development in multi-threaded embedded environments","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Agile software development; Computer science; Embedded system; Test (biology); Operating system; Computer architecture; Software engineering; Geology","score_opus":0.05255001405290805,"score_gpt":0.2996759811936519,"score_spread":0.24712596714074386,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1993945834","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18856741,0.00040426437,0.7919489,0.00026058863,0.0002047261,0.0007731663,0.00008870386,0.012550714,0.0052014547],"genre_scores_gemma":[0.59381706,0.00009933737,0.40131953,0.00015793974,0.000036704914,0.00026775192,0.000116326926,0.000267296,0.0039180615],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995116,0.00010731663,0.000032501783,0.00007732573,0.00019358138,0.00007760571],"domain_scores_gemma":[0.99869174,0.00043322146,0.000161804,0.00022371604,0.00034238523,0.00014712877],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005785848,0.00044206527,0.00033078727,0.00039940496,0.00021119938,0.00042790058,0.0015545578,0.00048826722,0.0041438225],"category_scores_gemma":[0.0014007038,0.00024630706,0.00020148483,0.00026307977,0.00024942364,0.000591756,0.00048575024,0.0007321293,0.0006765343],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002145975,0.0008670188,0.0068848645,0.0005305707,0.00010624872,0.0015647135,0.00052154064,0.012861333,0.52055705,0.004754085,0.008783943,0.44042265],"study_design_scores_gemma":[0.0010879211,0.009184417,0.0101767825,0.00012920015,0.00018942103,0.004861676,0.00020051775,0.2929534,0.64172816,0.0011786576,0.038204066,0.00010574133],"about_ca_topic_score_codex":0.00045761748,"about_ca_topic_score_gemma":0.0005978269,"teacher_disagreement_score":0.0041438225,"about_ca_system_score_codex":0.00023949819,"about_ca_system_score_gemma":0.000548924,"threshold_uncertainty_score":0.013862491},"labels":[],"label_agreement":null},{"id":"W1997272604","doi":"10.4236/jsea.2013.610a006","title":"A Survey of Software Test Estimation Techniques","year":2013,"lang":"en","type":"article","venue":"Journal of Software Engineering and Applications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"","keywords":"Software reliability testing; Regression testing; Estimation; Computer science; Software; Reliability engineering; System integration testing; Software development; Verification and validation; Software construction; Software testing; Software engineering; Systems engineering; Engineering; Operating system; Operations management","score_opus":0.011629271299270256,"score_gpt":0.23709443368206978,"score_spread":0.22546516238279954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1997272604","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013382526,0.10929126,0.86777276,0.0009916178,0.00022580549,0.00020874673,0.00042749246,0.0019773198,0.0057224757],"genre_scores_gemma":[0.20793496,0.15160675,0.6312487,0.0006785925,0.0010231544,0.000495796,0.0021957532,0.0007205133,0.004095735],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9920122,0.002021628,0.00089590607,0.001001261,0.0038327405,0.00023632689],"domain_scores_gemma":[0.9656784,0.0243884,0.0020702772,0.0020502596,0.0056344303,0.00017824833],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006030861,0.0017788471,0.002085199,0.0075291255,0.00044749436,0.0018175227,0.002969787,0.0013218303,0.001724698],"category_scores_gemma":[0.037115954,0.0010042378,0.0014586631,0.008165006,0.0006858178,0.004052182,0.0010652331,0.0018609272,0.0010778501],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008942022,0.00011821583,0.006142718,0.0016884285,0.00010013654,0.00008521322,0.00011248351,0.018370325,0.002169653,0.010238473,0.0030450418,0.95783985],"study_design_scores_gemma":[0.00012126477,0.0012550364,0.03400483,0.007297025,0.0009806404,0.004886416,0.00070613984,0.637466,0.03597505,0.09753286,0.17933014,0.0004445349],"about_ca_topic_score_codex":0.0025160857,"about_ca_topic_score_gemma":0.0016485496,"teacher_disagreement_score":0.0075291255,"about_ca_system_score_codex":0.0009432017,"about_ca_system_score_gemma":0.001297122,"threshold_uncertainty_score":0.031894624},"labels":[],"label_agreement":null},{"id":"W1997386306","doi":"10.1145/2484313.2484372","title":"Fuzzing the ActionScript virtual machine","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Natural Science Foundation of China; China Postdoctoral Science Foundation; University of Ottawa","keywords":"Computer science; Fuzz testing; Programming language; Test suite; Compiler; Suite; JavaScript; Context (archaeology); Code coverage; Source code; Code (set theory); Virtual machine; Test case; Software; Machine learning","score_opus":0.013828792639697263,"score_gpt":0.22308164968322258,"score_spread":0.2092528570435253,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1997386306","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3234044,0.00016356516,0.64529574,0.0002110132,0.00008694282,0.00016657484,0.00042934762,0.027222943,0.0030195331],"genre_scores_gemma":[0.7821816,0.00007365132,0.21311529,0.00015732087,0.000014794178,0.00015009199,0.0006465299,0.0016153654,0.0020452088],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9987704,0.00030212066,0.00009697609,0.00026021217,0.00045896045,0.000111322515],"domain_scores_gemma":[0.9961971,0.0020788652,0.00032957044,0.0009134658,0.00040827403,0.0000728137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010684909,0.0006782351,0.0003903619,0.00058079266,0.00031629213,0.0006992928,0.0010389849,0.0005963899,0.0019192693],"category_scores_gemma":[0.007105321,0.0003227533,0.0005263025,0.00030089676,0.0010933492,0.0012106149,0.0007230169,0.0006287648,0.00037824106],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015733008,0.0002626519,0.0268885,0.00061213935,0.00019602763,0.002224934,0.0013535985,0.12803991,0.4536317,0.059242476,0.0077659236,0.31820896],"study_design_scores_gemma":[0.00006058008,0.00043759862,0.003861288,0.000077130775,0.00007995701,0.0010294456,0.00010142933,0.5197741,0.43937227,0.020642713,0.014483605,0.00007997014],"about_ca_topic_score_codex":0.0010450956,"about_ca_topic_score_gemma":0.001081808,"teacher_disagreement_score":0.0019192693,"about_ca_system_score_codex":0.00048912107,"about_ca_system_score_gemma":0.0007289217,"threshold_uncertainty_score":0.0064206123},"labels":[],"label_agreement":null},{"id":"W1997899845","doi":"10.1145/1291535.1291541","title":"Model-based regression test suite generation using dependence analysis","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Extended finite-state machine; Test suite; Computer science; Regression analysis; Finite-state machine; Set (abstract data type); Suite; Model-based testing; Test case; Algorithm; Programming language; Machine learning","score_opus":0.0783813480523774,"score_gpt":0.33227034257455085,"score_spread":0.25388899452217345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1997899845","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04425755,0.00004952293,0.94996256,0.00008247169,0.000010398444,0.00013187085,0.0001379444,0.004024155,0.0013435774],"genre_scores_gemma":[0.5080674,0.00007403215,0.4890803,0.0000630704,0.000010488422,0.00027360776,0.000972438,0.00046480703,0.0009937851],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981001,0.000735091,0.00011780281,0.00028359724,0.0006446483,0.00011883878],"domain_scores_gemma":[0.99548596,0.0026004647,0.00040358337,0.0006872973,0.000751746,0.00007086632],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012379716,0.00076910976,0.00056868565,0.0013868985,0.0002251399,0.00039037084,0.0010054613,0.00052449486,0.0016197792],"category_scores_gemma":[0.008682632,0.00041912816,0.00096076477,0.00056188,0.00037367322,0.0007864232,0.00076830323,0.00066437386,0.00031541765],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040988054,0.0005433427,0.010164191,0.00027605292,0.00013925735,0.00072377577,0.0002789606,0.4899936,0.06448215,0.024417652,0.004402116,0.40416896],"study_design_scores_gemma":[0.000033784145,0.00009520995,0.0005288183,0.000014596478,0.000026311523,0.00016223894,0.000012022468,0.9774595,0.015819032,0.004666855,0.0011681716,0.0000135057035],"about_ca_topic_score_codex":0.0017403987,"about_ca_topic_score_gemma":0.0015341108,"teacher_disagreement_score":0.0017403987,"about_ca_system_score_codex":0.0005736426,"about_ca_system_score_gemma":0.0008513526,"threshold_uncertainty_score":0.0065470934},"labels":[],"label_agreement":null},{"id":"W1998421080","doi":"10.1007/s10009-010-0164-8","title":"Towards an industrial grade IVE for Java and next generation research platform for JML","year":2010,"lang":"en","type":"article","venue":"International Journal on Software Tools for Technology Transfer","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Programming language; Java; Java Modeling Language; Compiler; Assertion; Eclipse; Java annotation; Real time Java; Software engineering","score_opus":0.26508377289896135,"score_gpt":0.4006144419160369,"score_spread":0.13553066901707556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998421080","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03950153,0.00042269423,0.87916607,0.00081215356,0.00039173305,0.00076719595,0.00030056646,0.045474205,0.03316386],"genre_scores_gemma":[0.15470691,0.00032049068,0.7920955,0.00069534424,0.00011458199,0.00039625773,0.0015652238,0.013128275,0.03697754],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9972121,0.00054865336,0.00023544706,0.0004057494,0.0011567646,0.00044118045],"domain_scores_gemma":[0.99611276,0.0005694098,0.00023161128,0.0013813183,0.0011718089,0.00053322816],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047137253,0.0008093289,0.0007441094,0.0016485834,0.00078260346,0.0037981365,0.004024137,0.00206582,0.015607432],"category_scores_gemma":[0.007813308,0.0009831571,0.0011471327,0.0005937687,0.0009298211,0.0049418337,0.0034014145,0.003745657,0.006193421],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014931414,0.0021960037,0.006353696,0.0010116583,0.00011920651,0.0007312083,0.0015052018,0.0061260844,0.32513744,0.096802756,0.031436246,0.52708745],"study_design_scores_gemma":[0.0008164031,0.0028508848,0.0070475773,0.00075314153,0.0002907332,0.0013892585,0.00045849336,0.10031533,0.33719802,0.039661795,0.5089486,0.00026978404],"about_ca_topic_score_codex":0.0015579046,"about_ca_topic_score_gemma":0.0019394959,"teacher_disagreement_score":0.015607432,"about_ca_system_score_codex":0.0008770353,"about_ca_system_score_gemma":0.00222696,"threshold_uncertainty_score":0.05221212},"labels":[],"label_agreement":null},{"id":"W1998661399","doi":"10.1145/505776.505782","title":"Summary report of the OOPSLA 2000 workshop on scenario-based round-trip engineering","year":2001,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Software engineering; State (computer science); Conjunction (astronomy); World Wide Web; Library science; Programming language","score_opus":0.021462812271053417,"score_gpt":0.24234626439310675,"score_spread":0.22088345212205332,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998661399","genre_codex":"other","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021913169,0.030267227,0.25742513,0.0730981,0.10409004,0.0046794442,0.018629983,0.008350728,0.4815461],"genre_scores_gemma":[0.07371028,0.042346124,0.091077685,0.0068910145,0.011972477,0.0027995433,0.062199812,0.004884232,0.7041187],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9971155,0.0008166654,0.00018529467,0.00036237616,0.0012388673,0.0002812918],"domain_scores_gemma":[0.9897396,0.0014838356,0.00025135858,0.0006930128,0.0057456885,0.0020865453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0058013876,0.001343465,0.0007076325,0.0018206397,0.0013282854,0.00528587,0.0014718538,0.0014311955,0.10075352],"category_scores_gemma":[0.010505109,0.0005930583,0.0008052749,0.0016345053,0.00034200345,0.0036706687,0.0033568288,0.002370395,0.03309987],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002593356,0.00021692322,0.00033927147,0.00040334,0.000023215462,0.00016218114,0.00055596355,0.0020448302,0.0014994863,0.005224994,0.82084924,0.16842124],"study_design_scores_gemma":[0.000039931096,0.00010618489,0.0006135739,0.0003185463,0.00001842969,0.00007076144,0.00039606466,0.0012513582,0.0009381429,0.0021850034,0.9940316,0.00003052035],"about_ca_topic_score_codex":0.006518125,"about_ca_topic_score_gemma":0.006791241,"teacher_disagreement_score":0.10075352,"about_ca_system_score_codex":0.0020457201,"about_ca_system_score_gemma":0.00320227,"threshold_uncertainty_score":0.33705407},"labels":[],"label_agreement":null},{"id":"W2000579079","doi":"10.1016/s0167-6423(99)00046-5","title":"A calculus of program adaptation and its applications","year":2000,"lang":"en","type":"article","venue":"Science of Computer Programming","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke; Université Laval","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Software engineering; Software development; Programming language; Reuse; Component-based software engineering; Software construction; Adaptation (eye); Software","score_opus":0.02702946033575281,"score_gpt":0.29345959545317046,"score_spread":0.26643013511741764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2000579079","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008103387,0.0017244173,0.96368974,0.0015941123,0.00031840525,0.000038229202,0.000087794986,0.0007602397,0.023683693],"genre_scores_gemma":[0.50864846,0.0036938917,0.465607,0.0012891626,0.00113939,0.0003401663,0.0002099823,0.00046420743,0.01860777],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99789715,0.0006517857,0.00015300568,0.00046604124,0.0006077895,0.00022428841],"domain_scores_gemma":[0.9952696,0.0028491244,0.00027555798,0.00082573196,0.00053590164,0.00024403725],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029438876,0.0011306577,0.0011033982,0.002258033,0.0025372982,0.0037575176,0.002402404,0.0024129166,0.0049315346],"category_scores_gemma":[0.011086078,0.00097561395,0.0020644285,0.0034620329,0.008414135,0.007836757,0.0041265544,0.0059341257,0.0009489742],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008094821,0.000008399986,0.00006918294,0.000020345431,0.0000046602368,0.000060014958,0.00013875669,0.001490075,0.0002615238,0.9896344,0.00075163576,0.0075529106],"study_design_scores_gemma":[0.0000072330076,0.0000062498693,0.000056356377,0.000012085939,0.00001070476,0.00007295645,0.000023020697,0.010287312,0.00020906572,0.9839871,0.0053168726,0.0000110469355],"about_ca_topic_score_codex":0.0035873842,"about_ca_topic_score_gemma":0.0019436186,"teacher_disagreement_score":0.0049315346,"about_ca_system_score_codex":0.0023942569,"about_ca_system_score_gemma":0.0014798816,"threshold_uncertainty_score":0.017371655},"labels":[],"label_agreement":null},{"id":"W2002136372","doi":"10.1145/1363686.1363857","title":"An approach for supporting system-level test scenarios generation from textual use cases","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Tree traversal; Control flow; Finite-state machine; Test (biology); Scenario testing; Test case; State (computer science); Code coverage; Programming language; Theoretical computer science; Artificial intelligence; Machine learning; Software","score_opus":0.16133295948694185,"score_gpt":0.3028323202864848,"score_spread":0.14149936079954298,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002136372","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004060835,0.0000316174,0.98768276,0.0001393488,0.000014743897,0.00037066435,0.00022502575,0.0063993074,0.0010756489],"genre_scores_gemma":[0.06354005,0.00006335882,0.93286204,0.0001443656,0.000020167108,0.00065889314,0.0011844177,0.000681521,0.00084518106],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99475324,0.0021632328,0.00056648167,0.0007247083,0.0016246581,0.00016756971],"domain_scores_gemma":[0.9833835,0.010213572,0.0010968215,0.002861776,0.0021750282,0.0002692997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004073773,0.0013313999,0.00064811314,0.0024486668,0.0006164329,0.0021647462,0.0023229634,0.0017067865,0.0047266707],"category_scores_gemma":[0.023579435,0.0009547622,0.0014360463,0.00097525516,0.0013101253,0.0024308318,0.0019506299,0.0021610167,0.001704797],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061394105,0.0013079285,0.005638788,0.001709986,0.00028166233,0.0026962678,0.0032873384,0.07978686,0.090759665,0.071703896,0.013728091,0.7284855],"study_design_scores_gemma":[0.00032732292,0.0004984111,0.0015459607,0.00046296537,0.00022491408,0.0021872085,0.0003256444,0.75586945,0.11792574,0.05434135,0.06610608,0.00018498334],"about_ca_topic_score_codex":0.0013510038,"about_ca_topic_score_gemma":0.0021342207,"teacher_disagreement_score":0.0047266707,"about_ca_system_score_codex":0.00071541173,"about_ca_system_score_gemma":0.0013258385,"threshold_uncertainty_score":0.021544397},"labels":[],"label_agreement":null},{"id":"W2002702549","doi":"10.1017/s0960129513000054","title":"Verification of tree-processing programs via higher-order mode checking","year":2014,"lang":"en","type":"article","venue":"Mathematical Structures in Computer Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Career Trek","funders":"","keywords":"Soundness; Computer science; Programmer; Tree (set theory); Recursion (computer science); Programming language; Data structure; Algorithm; Theoretical computer science; Mathematics","score_opus":0.023457602234503876,"score_gpt":0.29050006344815543,"score_spread":0.26704246121365155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002702549","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009346327,0.000017399714,0.9886591,0.0000314796,0.000013520499,0.000030678908,0.00002393764,0.0015653003,0.0003123658],"genre_scores_gemma":[0.33015695,0.000090032576,0.6677332,0.000093981536,0.000028161434,0.00016896409,0.0001372702,0.00043743776,0.0011540572],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968027,0.0007044297,0.00024166337,0.00051534525,0.0014522838,0.00028348854],"domain_scores_gemma":[0.99173427,0.003982901,0.00069604174,0.0024710498,0.0009857619,0.0001300833],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034847734,0.0007737269,0.00063936174,0.0009920804,0.00056328136,0.0012168927,0.0026226556,0.0010354838,0.0029409132],"category_scores_gemma":[0.0075600273,0.00064025476,0.0016356916,0.0006089401,0.0022429693,0.0042466815,0.0019024016,0.0022793727,0.000469407],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063313625,0.0003747121,0.0056764144,0.00058364857,0.00022222033,0.000756433,0.0012397104,0.13882968,0.21795225,0.36105084,0.0028244834,0.26985648],"study_design_scores_gemma":[0.00013619378,0.00028286368,0.00066040613,0.00005555153,0.000096488395,0.00033880284,0.00005197068,0.75261503,0.14854577,0.09164704,0.005482525,0.0000873859],"about_ca_topic_score_codex":0.0021533014,"about_ca_topic_score_gemma":0.0025263052,"teacher_disagreement_score":0.0034847734,"about_ca_system_score_codex":0.00090607314,"about_ca_system_score_gemma":0.0019913958,"threshold_uncertainty_score":0.018429458},"labels":[],"label_agreement":null},{"id":"W2004981790","doi":"10.1016/j.dam.2010.12.022","title":"On a labeling problem in graphs","year":2011,"lang":"en","type":"article","venue":"Discrete Applied Mathematics","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Indian Institute of Technology Roorkee","keywords":"Combinatorics; Mathematics; Vertex (graph theory); Time complexity; Graph; Alphabet; Discrete mathematics","score_opus":0.03362565585068039,"score_gpt":0.24749419453963428,"score_spread":0.2138685386889539,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2004981790","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12976234,0.0009969901,0.81563914,0.017622944,0.00039335774,0.0002441196,0.0010799007,0.0007127015,0.033548564],"genre_scores_gemma":[0.5138327,0.002460679,0.44726625,0.0028211502,0.0010575086,0.00052926765,0.0037454735,0.0009875399,0.027299453],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99732804,0.001277845,0.00013649899,0.0006547688,0.00035053116,0.00025239412],"domain_scores_gemma":[0.96572435,0.029465294,0.0010480302,0.0018737216,0.00089557073,0.0009929774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031963396,0.0016527498,0.002359069,0.0021259421,0.0031148824,0.0045909556,0.0035801325,0.005847737,0.011244593],"category_scores_gemma":[0.023244968,0.0015363371,0.0022881213,0.0044844127,0.004792865,0.019745585,0.0040248055,0.0064560254,0.0010282829],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000259569,0.00029734173,0.0011378778,0.00041155235,0.000060800965,0.00021499068,0.0006884762,0.033807784,0.001083622,0.8958797,0.017307779,0.048850477],"study_design_scores_gemma":[0.00006057035,0.00003294783,0.00019939525,0.00006621458,0.00003431503,0.0000871518,0.0001650178,0.05638936,0.00045007537,0.9392503,0.0032489812,0.000015676203],"about_ca_topic_score_codex":0.0035724654,"about_ca_topic_score_gemma":0.002606462,"teacher_disagreement_score":0.011244593,"about_ca_system_score_codex":0.0028365895,"about_ca_system_score_gemma":0.0017049043,"threshold_uncertainty_score":0.03761691},"labels":[],"label_agreement":null},{"id":"W2005595981","doi":"10.1109/icstw.2013.41","title":"Experimenting with Category Partition's 1-Way and 2-Way Test Selection Criteria","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Intuition; Computer science; Selection (genetic algorithm); Test (biology); Partition (number theory); Test case; Machine learning; Artificial intelligence; Mathematics; Psychology","score_opus":0.015210897683390385,"score_gpt":0.24812051166885588,"score_spread":0.2329096139854655,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2005595981","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84251785,0.00026453123,0.15018503,0.00032985632,0.00006884737,0.00055952335,0.00024889474,0.0018918946,0.003933678],"genre_scores_gemma":[0.81320006,0.00007715017,0.18391295,0.00015843946,0.000023471717,0.0005982988,0.0004316056,0.00041665384,0.0011813982],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98643166,0.008032004,0.0008206801,0.001116176,0.0028266993,0.00077275],"domain_scores_gemma":[0.85236657,0.13006313,0.0035669012,0.007408582,0.005009146,0.0015856251],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0131425485,0.0013497588,0.00094291236,0.0022620137,0.00052716525,0.0015365734,0.0019255099,0.0017545884,0.0017300113],"category_scores_gemma":[0.071570516,0.0006104951,0.00084636686,0.0014413696,0.0011604925,0.0025599818,0.0021415735,0.0017679848,0.00038439062],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.011323287,0.00393652,0.044325363,0.001478096,0.00065217534,0.0008056697,0.0031375517,0.20334633,0.08912349,0.027447088,0.0050624725,0.60936195],"study_design_scores_gemma":[0.0017722205,0.012494431,0.029087601,0.00016845619,0.00045316305,0.0010603081,0.0012500866,0.81322867,0.108124815,0.022935966,0.009059268,0.00036504777],"about_ca_topic_score_codex":0.0024773874,"about_ca_topic_score_gemma":0.003029832,"teacher_disagreement_score":0.0131425485,"about_ca_system_score_codex":0.00109762,"about_ca_system_score_gemma":0.0012173946,"threshold_uncertainty_score":0.069505215},"labels":[],"label_agreement":null},{"id":"W2006253492","doi":"10.1016/j.jss.2007.05.037","title":"Traffic-aware stress testing of distributed real-time systems based on UML models using genetic algorithms","year":2007,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of Calgary","funders":"","keywords":"Computer science; Unified Modeling Language; Real-time computing; Test case; Focus (optics); Sequence diagram; Scenario testing; Algorithm; Stress test; Distributed computing; Artificial intelligence; Machine learning; Software; Variety (cybernetics)","score_opus":0.03951929322770974,"score_gpt":0.26828445766643977,"score_spread":0.22876516443873002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2006253492","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.532086,0.0001760833,0.46329358,0.00022262028,0.000031679658,0.00006273134,0.00004798339,0.0020787779,0.0020005985],"genre_scores_gemma":[0.9615555,0.000026556037,0.038052384,0.000015345451,0.000003850662,0.000026029213,0.000032479224,0.000043854925,0.00024403971],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99929285,0.00032825093,0.000028721937,0.0000887116,0.00018176087,0.00007962968],"domain_scores_gemma":[0.994311,0.0042266976,0.0005018054,0.00029756318,0.0005453008,0.000117691714],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009886987,0.0008205864,0.00066433905,0.0011266769,0.00037395622,0.00079813116,0.0011010069,0.00080622034,0.0007943844],"category_scores_gemma":[0.005626062,0.00035579322,0.000663427,0.00039194294,0.0005984919,0.00107208,0.0004937393,0.0004739964,0.00006945836],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018267013,0.00013253462,0.0032368433,0.000043165714,0.000052901658,0.0000921024,0.00011374071,0.9544748,0.007009468,0.0035842517,0.00017608644,0.03090157],"study_design_scores_gemma":[0.000009068759,0.00001672441,0.0001468807,0.000003002817,0.00001064183,0.000007441363,0.00000650197,0.99754626,0.0012248765,0.0009962379,0.000030113244,0.0000022053548],"about_ca_topic_score_codex":0.0074473997,"about_ca_topic_score_gemma":0.0062525165,"teacher_disagreement_score":0.0074473997,"about_ca_system_score_codex":0.0011332774,"about_ca_system_score_gemma":0.0010106951,"threshold_uncertainty_score":0.014808059},"labels":[],"label_agreement":null},{"id":"W2006575153","doi":"10.1109/icst.2013.23","title":"Efficient JavaScript Mutation Testing","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Mutation testing; JavaScript; Test suite; Mutation; Programming language; Web application; Process (computing); Set (abstract data type); Focus (optics); Test case; Software engineering; Data mining; Machine learning; Operating system","score_opus":0.0277027010756055,"score_gpt":0.23972739194392326,"score_spread":0.21202469086831777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2006575153","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13708907,0.00059918576,0.8381879,0.0003932771,0.000056267785,0.00032814374,0.00036545473,0.017081773,0.0058988784],"genre_scores_gemma":[0.5846514,0.00023819094,0.41125852,0.00016366752,0.000027002628,0.00017621534,0.0006312663,0.0008656872,0.0019880882],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.994125,0.0017467453,0.00032621628,0.0007572,0.0027222764,0.00032246404],"domain_scores_gemma":[0.9896863,0.0060322843,0.0010477756,0.0016482583,0.0013925036,0.00019278408],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021964817,0.0011104039,0.0009779815,0.0016963984,0.00048259183,0.0011294052,0.002038595,0.0010339302,0.0020943068],"category_scores_gemma":[0.015378578,0.00047149652,0.0007637008,0.0011591928,0.00080607855,0.0018628768,0.0012643876,0.00088997063,0.0009326178],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047460053,0.0005604109,0.013460289,0.00045699335,0.00014141854,0.0010087936,0.00028230742,0.18126026,0.13641194,0.015910296,0.0063060243,0.64372665],"study_design_scores_gemma":[0.00012948454,0.00024529092,0.003921677,0.000064982734,0.000058168545,0.000752336,0.00007508769,0.9067142,0.0639255,0.018705238,0.005358753,0.000049317405],"about_ca_topic_score_codex":0.0020295128,"about_ca_topic_score_gemma":0.0027899367,"teacher_disagreement_score":0.0021964817,"about_ca_system_score_codex":0.00072067045,"about_ca_system_score_gemma":0.001517919,"threshold_uncertainty_score":0.01161623},"labels":[],"label_agreement":null},{"id":"W2007279641","doi":"10.5555/2662593.2662596","title":"Towards model checking of computer games with Java PathFinder","year":2013,"lang":"en","type":"article","venue":"Computer Games","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Java; Computer science; Pathfinder; Programming language; Model checking; Code (set theory); State (computer science); Space (punctuation); Source code; State space; Operating system; World Wide Web; Mathematics; Set (abstract data type)","score_opus":0.022824778103721067,"score_gpt":0.23745149531079765,"score_spread":0.21462671720707657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2007279641","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017356949,0.00008630537,0.97566074,0.00041954615,0.000052296207,0.000074229225,0.00006700816,0.0053278683,0.00095509156],"genre_scores_gemma":[0.33179876,0.00024746984,0.6635006,0.0005908101,0.00008115063,0.00026754054,0.00032592955,0.0014657881,0.0017218832],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9896588,0.003854806,0.0006059076,0.0016039062,0.0032273817,0.0010492564],"domain_scores_gemma":[0.9785695,0.014116624,0.0015961069,0.003663467,0.0016643176,0.00038998868],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061043883,0.0020527286,0.0013641729,0.002556975,0.00094315456,0.0030523357,0.0039736116,0.0020604578,0.002872106],"category_scores_gemma":[0.033302613,0.0016141252,0.0040677204,0.0014529992,0.005694475,0.0068818056,0.0054628383,0.006353511,0.00063612],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006108408,0.0005829209,0.007602287,0.0006935798,0.00034496532,0.0013995006,0.0013180434,0.41476044,0.025255768,0.4090961,0.008169943,0.13016564],"study_design_scores_gemma":[0.000120488534,0.00012598366,0.0003445248,0.000112307665,0.00008322963,0.00024131894,0.00005836472,0.7785234,0.020195331,0.19518007,0.004941941,0.00007299181],"about_ca_topic_score_codex":0.010725291,"about_ca_topic_score_gemma":0.008951206,"teacher_disagreement_score":0.010725291,"about_ca_system_score_codex":0.002390241,"about_ca_system_score_gemma":0.0029632614,"threshold_uncertainty_score":0.032283485},"labels":[],"label_agreement":null},{"id":"W2008105683","doi":"10.2200/s00587ed1v01y201407swe002","title":"Hard Problems in Software Testing: Solutions Using Testing as a Service (TaaS)","year":2014,"lang":"en","type":"article","venue":"Synthesis lectures on software engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lockheed Martin (Canada)","funders":"","keywords":"Software testing; Computer science; Software; Service (business); Reliability engineering; Operating system; Engineering; Business","score_opus":0.0546395552355538,"score_gpt":0.2417101058015125,"score_spread":0.1870705505659587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2008105683","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012317035,0.0071188924,0.9515082,0.012607201,0.0016622068,0.00008718631,0.000081771126,0.0021868246,0.012430692],"genre_scores_gemma":[0.3636017,0.010221006,0.5951758,0.0024764398,0.0028781092,0.00024274089,0.00034430542,0.0009589942,0.024100915],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9976132,0.00072717323,0.00016188626,0.00031446465,0.00093289535,0.00025050156],"domain_scores_gemma":[0.99183816,0.005602789,0.00038805962,0.0006459103,0.0010332873,0.0004918425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028017382,0.001406787,0.001249213,0.0014191598,0.0010492207,0.003445233,0.0017523827,0.0022305993,0.008029779],"category_scores_gemma":[0.014048039,0.0005582639,0.0011188868,0.0022154471,0.002492602,0.0049748453,0.0027841337,0.005390741,0.001610996],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024537026,0.00022418756,0.0013129947,0.00072344625,0.00009651433,0.00026823272,0.00033951394,0.042009983,0.0051993085,0.20151205,0.042359527,0.7057088],"study_design_scores_gemma":[0.00009453066,0.00017677518,0.0006363801,0.0002255434,0.00007158634,0.0006582465,0.00029996232,0.30104578,0.0059233904,0.6429836,0.047825105,0.000059080678],"about_ca_topic_score_codex":0.0019506494,"about_ca_topic_score_gemma":0.0018476484,"teacher_disagreement_score":0.008029779,"about_ca_system_score_codex":0.001336532,"about_ca_system_score_gemma":0.0017383891,"threshold_uncertainty_score":0.026862323},"labels":[],"label_agreement":null},{"id":"W2012086719","doi":"10.1145/1022494.1022541","title":"Empirical studies of software testing techniques","year":2004,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Empirical research; Software testing; Software reliability testing; Software engineering; Regression testing; Test strategy; Software; Software construction; System integration testing; Software performance testing; Software development; Programming language","score_opus":0.07925087812671783,"score_gpt":0.3242008907103558,"score_spread":0.24495001258363797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2012086719","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8611504,0.024134219,0.0472495,0.007505211,0.00032528647,0.0005344324,0.0010586032,0.0001096088,0.0579327],"genre_scores_gemma":[0.9745874,0.00833524,0.012823884,0.0009953652,0.00024153996,0.0003232575,0.0007178183,0.00007340423,0.0019020699],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.94546115,0.03538168,0.0042394875,0.0023123515,0.011688688,0.0009166244],"domain_scores_gemma":[0.21581188,0.7080641,0.032231614,0.014020369,0.02839542,0.0014766146],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03494538,0.00052316545,0.00043849347,0.005149693,0.0009314025,0.0023108448,0.0018264749,0.0011573425,0.004283499],"category_scores_gemma":[0.36782643,0.00046979488,0.00039420885,0.006933367,0.0026664864,0.0059386366,0.0018593946,0.002674097,0.0007340473],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005303753,0.0034898773,0.57233745,0.004114528,0.0006939192,0.0004026322,0.0198199,0.0038527546,0.0019455303,0.06700066,0.0076008374,0.31821162],"study_design_scores_gemma":[0.00023309312,0.003506568,0.79398274,0.009879874,0.0005801651,0.0023192686,0.037710845,0.016921386,0.008162175,0.047051627,0.079465434,0.00018682158],"about_ca_topic_score_codex":0.0017679057,"about_ca_topic_score_gemma":0.0020006876,"teacher_disagreement_score":0.96505463,"about_ca_system_score_codex":0.0011159016,"about_ca_system_score_gemma":0.0014165619,"threshold_uncertainty_score":0.18481106},"labels":[],"label_agreement":null},{"id":"W2013579017","doi":"10.1109/coase.2010.5584605","title":"A Quality Framework to check the applicability of engineering and statistical assumptions for automated gauges","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Statistical process control; Correctness; Outlier; Computer science; Process (computing); Process capability; Automotive industry; Quality (philosophy); Industrial engineering; Data mining; Statistical model; Gauge (firearms); Work in process; Engineering; Algorithm; Artificial intelligence","score_opus":0.025020753896220906,"score_gpt":0.3472391183472207,"score_spread":0.32221836445099983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2013579017","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007934482,0.00009826294,0.9884956,0.00017007905,0.000019835023,0.00017416252,0.00010300069,0.00219715,0.00080737873],"genre_scores_gemma":[0.27798352,0.00011483902,0.7200123,0.00013348751,0.000060361443,0.00037399598,0.0004752445,0.00038381718,0.00046237576],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.95643735,0.008929968,0.004704654,0.0051292917,0.022844154,0.0019545222],"domain_scores_gemma":[0.89924926,0.026868775,0.0175531,0.02081429,0.033537623,0.0019770006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.029358203,0.0018023247,0.001826994,0.007429142,0.0015584328,0.005739118,0.004017434,0.0021883787,0.0010677583],"category_scores_gemma":[0.0820252,0.0011020105,0.0026124737,0.0026342848,0.004469583,0.0063380445,0.0037134117,0.003737378,0.00041543692],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050883985,0.0006085029,0.031813785,0.00057202467,0.00037064246,0.00057896314,0.0011534747,0.39329535,0.02529131,0.29337743,0.0049518254,0.24747786],"study_design_scores_gemma":[0.00007334747,0.0006386526,0.00424329,0.00014590699,0.00012458372,0.00036030728,0.00016950964,0.92295563,0.017704004,0.046044834,0.007408599,0.0001314068],"about_ca_topic_score_codex":0.017008398,"about_ca_topic_score_gemma":0.006457105,"teacher_disagreement_score":0.029358203,"about_ca_system_score_codex":0.0041534286,"about_ca_system_score_gemma":0.007581441,"threshold_uncertainty_score":0.15526289},"labels":[],"label_agreement":null},{"id":"W2013851052","doi":"10.1007/s10664-006-9031-3","title":"A practical approach to testing GUI systems","year":2006,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Component (thermodynamics); Set (abstract data type); Graph; Graphical user interface; Software engineering; Programming language; Theoretical computer science","score_opus":0.060666448574408956,"score_gpt":0.297102360528753,"score_spread":0.23643591195434405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2013851052","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004894258,0.00016470095,0.9773457,0.003998331,0.00007409487,0.00032064237,0.00006687469,0.0009962765,0.012139055],"genre_scores_gemma":[0.100347,0.00030690272,0.88969845,0.00074597035,0.000099883706,0.00069613894,0.00013973871,0.00014192302,0.007823979],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.987906,0.0064177066,0.00046127307,0.00095943204,0.0039385096,0.00031707613],"domain_scores_gemma":[0.96282357,0.0219869,0.0012269198,0.008175828,0.004870316,0.00091632834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00861635,0.0017118378,0.0009752092,0.0028573736,0.0019701368,0.003348725,0.004666108,0.003875441,0.021351121],"category_scores_gemma":[0.054907292,0.0012181255,0.000702248,0.0020249747,0.0037731796,0.0063316994,0.0050050262,0.0054374337,0.0042239535],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016457739,0.0012773499,0.0059736646,0.0006886883,0.00008033001,0.0008401395,0.0015688271,0.01230979,0.008586078,0.3409224,0.018601198,0.60898703],"study_design_scores_gemma":[0.0002987284,0.0008661661,0.004537025,0.00041697483,0.00008093722,0.0032851151,0.0019382264,0.09948977,0.0061980607,0.818778,0.06401217,0.00009884553],"about_ca_topic_score_codex":0.0014967447,"about_ca_topic_score_gemma":0.003424928,"teacher_disagreement_score":0.021351121,"about_ca_system_score_codex":0.0011344921,"about_ca_system_score_gemma":0.0026466008,"threshold_uncertainty_score":0.07142663},"labels":[],"label_agreement":null},{"id":"W2015146187","doi":"10.1109/ase.2013.6693121","title":"PYTHIA: Generating test cases with oracles for JavaScript applications","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"JavaScript; Computer science; Unobtrusive JavaScript; Unit testing; Programming language; Test case; Web application; Point (geometry); Code (set theory); Operating system; Rich Internet application; Software; Machine learning","score_opus":0.023159578142140497,"score_gpt":0.2505283463855822,"score_spread":0.22736876824344168,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2015146187","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07425164,0.00022418641,0.82006145,0.00020905063,0.000037018122,0.0005589062,0.0010950618,0.10105479,0.0025077919],"genre_scores_gemma":[0.37658513,0.000204361,0.60904175,0.000140801,0.000023179107,0.0006103377,0.0037716273,0.008125358,0.001497533],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975496,0.0008109891,0.00026244632,0.0003885554,0.0007776346,0.00021087068],"domain_scores_gemma":[0.98498803,0.010847067,0.0010397986,0.0017313069,0.0010968749,0.00029699414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031566145,0.0015151558,0.0006435335,0.002782491,0.0003918189,0.0013594022,0.0023067794,0.0012804052,0.0036671702],"category_scores_gemma":[0.023338687,0.00086829613,0.0010509974,0.0009739821,0.001490556,0.0016806728,0.0017354963,0.0012102016,0.0011213559],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011869434,0.000940408,0.035815626,0.0016529042,0.00021621009,0.004198581,0.001865264,0.1152503,0.09182515,0.019081084,0.023556815,0.70441073],"study_design_scores_gemma":[0.000374514,0.0006440742,0.006435116,0.00026774794,0.00010520592,0.0024779409,0.0002853432,0.7729891,0.18239576,0.016181689,0.017707884,0.00013571774],"about_ca_topic_score_codex":0.0021212052,"about_ca_topic_score_gemma":0.0018070181,"teacher_disagreement_score":0.0036671702,"about_ca_system_score_codex":0.0006334486,"about_ca_system_score_gemma":0.0010581436,"threshold_uncertainty_score":0.01669401},"labels":[],"label_agreement":null},{"id":"W2016250485","doi":"10.1016/s0140-3664(99)00227-3","title":"Test generation based on control and data dependencies within system specifications in SDL","year":2000,"lang":"en","type":"article","venue":"Computer Communications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Instituto de Telecomunicações","keywords":"Computer science; Control flow; Data flow diagram; Programming language; Specification language; Extended finite-state machine; System requirements specification; Control (management); Test data; Finite-state machine; Distributed computing; Software engineering; Database; Artificial intelligence","score_opus":0.13848095803885363,"score_gpt":0.2977731457244659,"score_spread":0.15929218768561226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2016250485","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1677621,0.000111963425,0.8226524,0.00028937522,0.00003433215,0.00017876898,0.00021781209,0.006470288,0.0022829692],"genre_scores_gemma":[0.8571351,0.000048121736,0.14105774,0.00015297259,0.000010816312,0.00013426867,0.0003003967,0.00045511217,0.0007054339],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99545395,0.0021850595,0.0003182813,0.00041892586,0.0012050534,0.00041871774],"domain_scores_gemma":[0.9724325,0.022616453,0.0012855586,0.0017397065,0.0016569403,0.0002689174],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031886706,0.00081381935,0.00056338875,0.0014311281,0.00037087436,0.0009482034,0.0010464506,0.0008738433,0.0021966405],"category_scores_gemma":[0.01633674,0.00061798265,0.000658533,0.00058487395,0.0011890857,0.0016336979,0.0011421639,0.0010183913,0.0002886934],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002947996,0.00067308656,0.017740443,0.0009228328,0.0002003602,0.002117965,0.001321008,0.42517167,0.08394592,0.06974691,0.0049726847,0.39023918],"study_design_scores_gemma":[0.00016371335,0.00024383704,0.0007531051,0.000053489566,0.000066613036,0.00021965132,0.000046990892,0.9164503,0.06574154,0.015185737,0.001049583,0.000025436479],"about_ca_topic_score_codex":0.0027027645,"about_ca_topic_score_gemma":0.0039662523,"teacher_disagreement_score":0.0031886706,"about_ca_system_score_codex":0.0008907299,"about_ca_system_score_gemma":0.0013345114,"threshold_uncertainty_score":0.016863525},"labels":[],"label_agreement":null},{"id":"W2019724127","doi":"10.1109/icst.2012.205","title":"Incremental Test Case Generation for UML-RT Models Using Symbolic Execution","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Unified Modeling Language; Test case; Software engineering; Leverage (statistics); Iterative and incremental development; Test Management Approach; Symbolic execution; Code coverage; Code generation; System under test; Software development; Software; Programming language; Software construction; Artificial intelligence; Machine learning; Operating system","score_opus":0.15395948223293526,"score_gpt":0.32926824706890745,"score_spread":0.1753087648359722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2019724127","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1034774,0.00009558656,0.88397324,0.00020989023,0.00003671726,0.0004502185,0.00039874655,0.008588983,0.0027692397],"genre_scores_gemma":[0.4361476,0.00012488541,0.5597439,0.0000721508,0.000014111689,0.00038524173,0.0013941532,0.00085294835,0.001265089],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974177,0.0010955032,0.00018029439,0.00026301216,0.0008791075,0.00016440776],"domain_scores_gemma":[0.98474437,0.011656609,0.0007427903,0.0016564633,0.0010287904,0.00017091323],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023111305,0.0012511571,0.0005523402,0.00225213,0.0003664697,0.0009890519,0.0017712573,0.0010124688,0.0040792404],"category_scores_gemma":[0.02033168,0.00061247667,0.0012187776,0.00080371555,0.00068470236,0.0015065022,0.001344039,0.00095406524,0.0006758805],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005565419,0.0006497147,0.007818971,0.0005190416,0.00012417193,0.0016443951,0.0011585256,0.63937753,0.05497203,0.019366447,0.003205475,0.2706071],"study_design_scores_gemma":[0.00007591626,0.00018950191,0.00041134606,0.00004625561,0.000035557387,0.00018493511,0.00007477999,0.96637505,0.025286652,0.005056072,0.0022402273,0.000023791461],"about_ca_topic_score_codex":0.004547156,"about_ca_topic_score_gemma":0.0067101913,"teacher_disagreement_score":0.004547156,"about_ca_system_score_codex":0.0009056398,"about_ca_system_score_gemma":0.001555539,"threshold_uncertainty_score":0.013646483},"labels":[],"label_agreement":null},{"id":"W2019797727","doi":"10.1016/j.jss.2009.05.019","title":"Exploring alternatives for transition verification","year":2009,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Sequence (biology); Invertible matrix; Finite-state machine; Computer science; Model checking; State (computer science); Range (aeronautics); Transition system; Automaton; Algorithm; Conformance testing; Theoretical computer science; Programming language; Mathematics; Engineering","score_opus":0.10664762297229999,"score_gpt":0.28679998609778135,"score_spread":0.18015236312548136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2019797727","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025397154,0.0005806484,0.9519058,0.0023676644,0.00025619176,0.000101832804,0.00009411781,0.001085193,0.018211499],"genre_scores_gemma":[0.61830777,0.00033942203,0.37586895,0.0003639153,0.0001398327,0.00013279129,0.00020530254,0.00035286817,0.0042892112],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9915496,0.004266397,0.00053610984,0.0011715379,0.001738966,0.0007372973],"domain_scores_gemma":[0.96858996,0.02209521,0.00077167863,0.005971681,0.002122507,0.0004490704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006481001,0.00093404856,0.0011923079,0.002039878,0.001898226,0.003697259,0.0028120512,0.0035051669,0.015298436],"category_scores_gemma":[0.04012554,0.0009379145,0.002656352,0.0016743428,0.0053999936,0.0114549715,0.0045348895,0.004302786,0.0017710741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004402744,0.00011711943,0.00077393424,0.00018920143,0.000050978757,0.00021214204,0.00047390372,0.016065497,0.0012762046,0.88060564,0.0024858525,0.097309195],"study_design_scores_gemma":[0.00005943529,0.000082589846,0.00008663699,0.00007350998,0.000049370858,0.000077548175,0.00017521244,0.09247457,0.001409821,0.9015407,0.003942947,0.000027723083],"about_ca_topic_score_codex":0.0014418447,"about_ca_topic_score_gemma":0.00262129,"teacher_disagreement_score":0.015298436,"about_ca_system_score_codex":0.0011567512,"about_ca_system_score_gemma":0.0017402148,"threshold_uncertainty_score":0.051178336},"labels":[],"label_agreement":null},{"id":"W2020104929","doi":"10.1145/2591062.2591168","title":"A compiler project with learning progressions","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Compiler; Computer science; Compiler construction; Structuring; Programming language; Optimizing compiler; Compiler correctness; Course (navigation); Software engineering; Mathematics education; Engineering; Psychology","score_opus":0.023419842353767893,"score_gpt":0.27901821184150116,"score_spread":0.2555983694877333,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2020104929","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54854274,0.00040637687,0.3439434,0.0047950353,0.0014478391,0.002132635,0.00059351855,0.011833525,0.08630494],"genre_scores_gemma":[0.5024687,0.0002836638,0.4116325,0.00078819465,0.00015467324,0.0010390684,0.0009101755,0.0011461262,0.081576854],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981383,0.0006477984,0.00006733495,0.00031615663,0.0004981708,0.00033223076],"domain_scores_gemma":[0.9953844,0.00050401874,0.00013092067,0.0004383865,0.0010419592,0.002500362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029906046,0.0005628601,0.0003537415,0.0007616299,0.0023153685,0.0024118253,0.0010438449,0.0013479181,0.011413555],"category_scores_gemma":[0.0063538607,0.00034249108,0.0005047421,0.0006206843,0.0010771736,0.001985695,0.0034151378,0.0025128496,0.0031116533],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014980367,0.010011022,0.01307941,0.00039730206,0.00003434602,0.0020028003,0.014436263,0.01233453,0.039761066,0.10810275,0.074037924,0.72430456],"study_design_scores_gemma":[0.001301669,0.018539438,0.013451257,0.00045347074,0.00007540314,0.0048578097,0.009376331,0.056072608,0.09729064,0.06526905,0.7329495,0.00036279488],"about_ca_topic_score_codex":0.0013038564,"about_ca_topic_score_gemma":0.0021114394,"teacher_disagreement_score":0.011413555,"about_ca_system_score_codex":0.0010785589,"about_ca_system_score_gemma":0.0047233407,"threshold_uncertainty_score":0.03818214},"labels":[],"label_agreement":null},{"id":"W2021654859","doi":"10.1109/asqed.2013.6643604","title":"Mu-GSIM: A mutation testing simulator on GPUs","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; CMC Microsystems (Canada)","funders":"","keywords":"Computer science; Parallel computing; Speedup; Leverage (statistics); Kernel (algebra); Graphics; CUDA; Computer architecture; Operating system","score_opus":0.031850754151639356,"score_gpt":0.2596742175700641,"score_spread":0.22782346341842474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021654859","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26375425,0.00032543115,0.68620414,0.00047164608,0.00019728919,0.0002082486,0.0014515092,0.025979863,0.021407593],"genre_scores_gemma":[0.7770942,0.00021119797,0.21468087,0.00016637257,0.000018765439,0.00025060272,0.0011221803,0.0014804759,0.0049752113],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997601,0.00005933756,0.00001332844,0.00002377434,0.00011393686,0.000029554],"domain_scores_gemma":[0.9995042,0.00021835443,0.000048903985,0.00006553445,0.00012330603,0.000039736446],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00034460952,0.00051602285,0.00045267405,0.00050218013,0.00030432997,0.0004358933,0.0015074593,0.0006510916,0.0035118503],"category_scores_gemma":[0.0018501459,0.00032443277,0.0005075519,0.00048390598,0.00040842997,0.00059513556,0.00050088536,0.0007707227,0.00042306288],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020823271,0.00009304691,0.0034531483,0.00011650558,0.00007274287,0.0002449041,0.0001178553,0.9164693,0.026565135,0.011115637,0.006223719,0.035319787],"study_design_scores_gemma":[0.000020040488,0.000027951895,0.00016885887,0.0000032977164,0.000004857528,0.000022200682,0.0000055347364,0.9913127,0.00549283,0.00085584365,0.0020807455,0.0000050566596],"about_ca_topic_score_codex":0.0070313914,"about_ca_topic_score_gemma":0.006664959,"teacher_disagreement_score":0.0070313914,"about_ca_system_score_codex":0.0006657613,"about_ca_system_score_gemma":0.0012605559,"threshold_uncertainty_score":0.013980925},"labels":[],"label_agreement":null},{"id":"W2022255738","doi":"10.1109/ccece.2008.4564528","title":"Verification of program dynamic behaviours based on static analysis","year":2008,"lang":"en","type":"article","venue":"Conference proceedings - Canadian Conference on Electrical and Computer Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Static analysis; Coding (social sciences); Program analysis; Dynamic program analysis; Static program analysis; Source code; Dynamic programming; State (computer science); Code (set theory); Programming language; Algorithm; Software; Software development","score_opus":0.02033448777341849,"score_gpt":0.2336014875384555,"score_spread":0.21326699976503702,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2022255738","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08543289,0.000119534336,0.9012968,0.000087349916,0.000026979733,0.00017439478,0.00036814236,0.010256552,0.0022374324],"genre_scores_gemma":[0.7204944,0.00022577209,0.27491623,0.00008363716,0.000027544571,0.00043259616,0.0010562394,0.0011044738,0.0016590968],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949587,0.0011269321,0.00036948215,0.00071835687,0.0024541214,0.00037242388],"domain_scores_gemma":[0.9848909,0.0072514126,0.0014387458,0.0036598241,0.002624894,0.0001343344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026136094,0.0012275027,0.0009845949,0.0028201498,0.00084368355,0.001223852,0.0013511794,0.0008203474,0.0021296842],"category_scores_gemma":[0.012964008,0.00080825685,0.0016069555,0.0010330235,0.0017102087,0.0027512698,0.0011067226,0.001013967,0.00067251176],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010721446,0.000610078,0.022889208,0.0012998803,0.00043567223,0.0015718917,0.0011243729,0.27779788,0.34669942,0.057134196,0.0022115747,0.28715375],"study_design_scores_gemma":[0.000073728006,0.00045366838,0.0037650196,0.00013784508,0.00017206628,0.0005925617,0.00012913103,0.75994015,0.20963046,0.020218464,0.004790585,0.00009627866],"about_ca_topic_score_codex":0.0031108176,"about_ca_topic_score_gemma":0.0025442566,"teacher_disagreement_score":0.0031108176,"about_ca_system_score_codex":0.0010922275,"about_ca_system_score_gemma":0.002431883,"threshold_uncertainty_score":0.013822198},"labels":[],"label_agreement":null},{"id":"W2022887099","doi":"10.1145/1569901.1570123","title":"MC/DC automatic test input data generation","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Software quality; Computer science; Reliability engineering; Code coverage; Fitness function; Software; Random testing; Chaining; Test suite; Test data; Test case; Algorithm; Engineering; Software development; Machine learning; Software engineering; Genetic algorithm; Programming language","score_opus":0.07731818523370544,"score_gpt":0.30886928560908844,"score_spread":0.231551100375383,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2022887099","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13316862,0.00019651509,0.8555294,0.0001668225,0.000057469682,0.0002655737,0.00025305524,0.004350037,0.006012536],"genre_scores_gemma":[0.7367501,0.00006607958,0.25825906,0.00015752311,0.000024627108,0.00028793587,0.00079086336,0.00031695006,0.0033469622],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99914944,0.00021070792,0.000056702018,0.00015886543,0.00033165736,0.00009254236],"domain_scores_gemma":[0.99714077,0.0014209136,0.00017186721,0.00051262625,0.0006693477,0.00008448204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086633547,0.0006279842,0.00056341535,0.0008728348,0.0002890374,0.00057157903,0.0007764472,0.0006551045,0.0037738266],"category_scores_gemma":[0.004707807,0.00016888448,0.0003470588,0.00068485824,0.00038686697,0.0004809549,0.00067343406,0.0004707481,0.000592022],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054749363,0.00023708181,0.0060782926,0.00021481582,0.000049096463,0.00046650154,0.00014145505,0.2552996,0.059057955,0.014185829,0.0051584323,0.65856355],"study_design_scores_gemma":[0.00004245685,0.00012698384,0.0009896894,0.000014440257,0.000015024888,0.00023961632,0.000013325233,0.9560868,0.035604198,0.0036929513,0.0031616976,0.00001279614],"about_ca_topic_score_codex":0.0016853245,"about_ca_topic_score_gemma":0.0010919243,"teacher_disagreement_score":0.0037738266,"about_ca_system_score_codex":0.0005183798,"about_ca_system_score_gemma":0.0006979463,"threshold_uncertainty_score":0.012624741},"labels":[],"label_agreement":null},{"id":"W2023975198","doi":"10.1109/sera.2010.46","title":"Modeling and Validating Requirements Using Executable Cotnracts and Scenarios","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University; Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Executable; Software engineering; Functional requirement; Context (archaeology); Non-functional requirement; Set (abstract data type); Model-based testing; Test (biology); Key (lock); Conformance testing; Test case; Software; Software development; Programming language; Machine learning; Software construction; Standardization; Operating system","score_opus":0.05552498095438944,"score_gpt":0.30476977822824,"score_spread":0.24924479727385054,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2023975198","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09961069,0.0000745502,0.8829129,0.000380745,0.000030707735,0.001109053,0.0006774464,0.004438019,0.010766002],"genre_scores_gemma":[0.31487265,0.00019643358,0.67967856,0.00006877401,0.000010578002,0.0007257249,0.0010994301,0.0003571465,0.0029906677],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99607605,0.0021214273,0.00038200134,0.0003249649,0.00095254974,0.00014304723],"domain_scores_gemma":[0.9906573,0.005011939,0.00078061904,0.0022626326,0.0010941637,0.000193269],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045997566,0.0013034361,0.00037892137,0.0015556397,0.00053244695,0.0022512334,0.0018938115,0.0015546511,0.0033667148],"category_scores_gemma":[0.011512062,0.0008282354,0.0011514801,0.0006152006,0.00103734,0.0023457143,0.0014256142,0.001115859,0.0005852629],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035105488,0.00067116466,0.00929721,0.0005315718,0.00009930737,0.0024527926,0.003700205,0.70470405,0.021703904,0.16683127,0.0025010384,0.08715649],"study_design_scores_gemma":[0.00007135265,0.00022553004,0.0010151411,0.00016584372,0.00003661529,0.0004289533,0.00039581055,0.9414565,0.012284024,0.025955092,0.01791579,0.00004928653],"about_ca_topic_score_codex":0.0052740616,"about_ca_topic_score_gemma":0.008212128,"teacher_disagreement_score":0.0052740616,"about_ca_system_score_codex":0.0012265438,"about_ca_system_score_gemma":0.0015754424,"threshold_uncertainty_score":0.024326086},"labels":[],"label_agreement":null},{"id":"W2024965224","doi":"10.4236/jsea.2013.610a005","title":"Traceability in Acceptance Testing","year":2013,"lang":"en","type":"article","venue":"Journal of Software Engineering and Applications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University; Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Traceability; Computer science; Software engineering; Acceptance testing; Requirements traceability; Executable; Quality (philosophy); Process (computing); Stakeholder; Model-based testing; Systems engineering; Test strategy; Risk analysis (engineering); Software; Reliability engineering; Test case; Software development; Engineering; Requirement; Programming language","score_opus":0.012713276665239758,"score_gpt":0.2267064411533581,"score_spread":0.21399316448811836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2024965224","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019809581,0.0019480725,0.96092325,0.0021665974,0.00013757552,0.00027189837,0.000064981614,0.000562586,0.014115545],"genre_scores_gemma":[0.7107495,0.0031330693,0.27627185,0.0013038947,0.0004724105,0.0012223261,0.00033891175,0.00042461976,0.0060833776],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9547568,0.026235154,0.002935203,0.003859118,0.010965261,0.0012484342],"domain_scores_gemma":[0.7568335,0.20072554,0.009058879,0.022188984,0.010040355,0.0011526335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.032041322,0.0017348166,0.0015852592,0.0046519185,0.0015341566,0.0050655515,0.00425943,0.0052847518,0.0051782336],"category_scores_gemma":[0.19124141,0.0011502195,0.0019915039,0.0029775922,0.010145007,0.014601612,0.0039354484,0.005462732,0.00067150756],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017251499,0.0002563039,0.0062312926,0.0007336629,0.00012386602,0.00095116044,0.0025200455,0.037776567,0.0017427218,0.80559605,0.0010087977,0.14288698],"study_design_scores_gemma":[0.00008768563,0.00031592773,0.0012195732,0.00054407463,0.00008192646,0.0008952091,0.00040537305,0.08800304,0.0030402907,0.89596224,0.009370129,0.00007441874],"about_ca_topic_score_codex":0.0041081323,"about_ca_topic_score_gemma":0.0013954233,"teacher_disagreement_score":0.032041322,"about_ca_system_score_codex":0.002748452,"about_ca_system_score_gemma":0.0029020205,"threshold_uncertainty_score":0.16945273},"labels":[],"label_agreement":null},{"id":"W2025618033","doi":"10.1007/s10270-002-0004-8","title":"A UML-Based Approach to System Testing","year":2002,"lang":"en","type":"article","venue":"Software & Systems Modeling","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":303,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Unified Modeling Language; Sequence diagram; Class diagram; Object Constraint Language; Software engineering; Applications of UML; Testability; Model-based testing; Programming language; UML tool; Context (archaeology); Test case; Test Management Approach; Systems Modeling Language; Activity diagram; System testing; Software development; Reliability engineering; Software; Software construction","score_opus":0.09020794658056883,"score_gpt":0.23924228661756475,"score_spread":0.14903434003699592,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2025618033","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008865838,0.00014151701,0.9941871,0.0003244034,0.00004083728,0.00006556288,0.00002652638,0.0009066582,0.0034207378],"genre_scores_gemma":[0.051211227,0.0003507326,0.9442317,0.00029585356,0.00007275666,0.0001947028,0.00015004887,0.00033551131,0.0031574618],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9917076,0.0035017417,0.0005244516,0.0004615456,0.00353784,0.00026677919],"domain_scores_gemma":[0.99087137,0.0048828493,0.00037015523,0.0020897086,0.0015608786,0.0002250476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005754771,0.0011409867,0.00083144783,0.0032379662,0.0010371004,0.0034313796,0.0032840506,0.0019970092,0.0053855916],"category_scores_gemma":[0.01883765,0.001087818,0.0016475619,0.0014893862,0.0022814472,0.0053011933,0.0025249429,0.003775695,0.0016036974],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007865132,0.00032912387,0.0010540042,0.00033714398,0.00010160663,0.000330659,0.0007536746,0.042095844,0.009260534,0.6227638,0.0074450416,0.31544995],"study_design_scores_gemma":[0.00008856354,0.00014779656,0.00067070586,0.00044707267,0.00016927061,0.000826291,0.00015577664,0.4362407,0.013475786,0.47919646,0.06850389,0.000077723336],"about_ca_topic_score_codex":0.003171894,"about_ca_topic_score_gemma":0.0046499404,"teacher_disagreement_score":0.005754771,"about_ca_system_score_codex":0.0013859853,"about_ca_system_score_gemma":0.0019593458,"threshold_uncertainty_score":0.03043449},"labels":[],"label_agreement":null},{"id":"W2026184507","doi":"10.1093/comjnl/bxm096","title":"The Effect of the Distributed Test Architecture on the Power of Testing","year":2007,"lang":"en","type":"article","venue":"The Computer Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Observability; Controllability; Finite-state machine; Computer science; Equivalence (formal languages); Conformance testing; Architecture; Mathematics; Algorithm; Discrete mathematics","score_opus":0.012032451917902767,"score_gpt":0.24116125434326105,"score_spread":0.22912880242535827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2026184507","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14074601,0.0013522032,0.8083158,0.00659925,0.00020855812,0.00019120601,0.00014607352,0.0018747477,0.040566124],"genre_scores_gemma":[0.8309551,0.00054563506,0.16273119,0.00060539023,0.00022846936,0.00020485175,0.000120492754,0.00029973747,0.004309133],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98749906,0.005357737,0.0005450567,0.0026550863,0.0033746993,0.0005684669],"domain_scores_gemma":[0.8898235,0.08440849,0.0024650823,0.018087067,0.004241142,0.000974735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0076257694,0.0013177564,0.0010609233,0.0009678589,0.0012153086,0.0026201895,0.0025364417,0.002239783,0.0058646156],"category_scores_gemma":[0.032555014,0.0009148127,0.0014253489,0.0010251162,0.0079826955,0.011402203,0.0044712336,0.007036853,0.0009151223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011767753,0.00032912235,0.006327148,0.0006186896,0.00014078945,0.0007159643,0.00085114694,0.21083722,0.02435666,0.5750541,0.0026635595,0.17692883],"study_design_scores_gemma":[0.00036743673,0.00068996963,0.002215872,0.00015810937,0.00015083507,0.00070405495,0.00025760196,0.45570964,0.026468126,0.5006284,0.012570273,0.000079690486],"about_ca_topic_score_codex":0.002877888,"about_ca_topic_score_gemma":0.0021681974,"teacher_disagreement_score":0.0076257694,"about_ca_system_score_codex":0.002208828,"about_ca_system_score_gemma":0.0020764521,"threshold_uncertainty_score":0.040329456},"labels":[],"label_agreement":null},{"id":"W2028661287","doi":"10.1016/j.ipl.2005.01.011","title":"Minimizing the number of inputs while applying adaptive test cases","year":2005,"lang":"en","type":"article","venue":"Information Processing Letters","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Computerized adaptive testing; Test (biology); Algorithm; Mathematics; Statistics","score_opus":0.024393448385554278,"score_gpt":0.26021796415042353,"score_spread":0.23582451576486926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2028661287","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24601549,0.00038197835,0.744472,0.00054195634,0.000052080726,0.0003605315,0.00011081623,0.0043400275,0.0037251262],"genre_scores_gemma":[0.67304856,0.00012259222,0.32466704,0.00019093898,0.00003595937,0.00014345924,0.00017554044,0.00039800245,0.0012179107],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9951881,0.0017636014,0.00042969923,0.0007389644,0.0014209376,0.00045873306],"domain_scores_gemma":[0.95745337,0.030633327,0.003276767,0.0048158905,0.0032599543,0.0005606019],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001684459,0.0020064437,0.0012166284,0.0012375702,0.00043868512,0.001151433,0.0029166427,0.0011071565,0.0027862035],"category_scores_gemma":[0.028121855,0.0005950519,0.0005631628,0.0010600444,0.0006439092,0.0024410181,0.0011626419,0.0016322542,0.00063054264],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024001964,0.00084891956,0.016921842,0.0005497448,0.00016553642,0.00063252676,0.00040937925,0.16416943,0.1286843,0.005553603,0.0024254723,0.6772391],"study_design_scores_gemma":[0.00021930784,0.0008980326,0.00793605,0.000117369564,0.00022378222,0.00079575623,0.00025710597,0.87898386,0.08916122,0.018465847,0.0028789656,0.00006272462],"about_ca_topic_score_codex":0.001756692,"about_ca_topic_score_gemma":0.0039202846,"teacher_disagreement_score":0.0029166427,"about_ca_system_score_codex":0.0007725853,"about_ca_system_score_gemma":0.0016240607,"threshold_uncertainty_score":0.009320736},"labels":[],"label_agreement":null},{"id":"W2029160271","doi":"10.1007/s11219-014-9232-8","title":"Toward a mature industrial practice of software test automation","year":2014,"lang":"en","type":"article","venue":"Software Quality Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Automation; Software engineering; Test (biology); Software; Computer science; Software testing; Engineering; Operating system","score_opus":0.06356689489556629,"score_gpt":0.33149529574143893,"score_spread":0.26792840084587266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2029160271","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016066948,0.0067835376,0.87433636,0.044877943,0.00079804234,0.00042027293,0.00006197855,0.0021802746,0.05447453],"genre_scores_gemma":[0.24724823,0.0061793304,0.72984844,0.0052776616,0.0011327298,0.00040129636,0.00017547031,0.00061259966,0.009124179],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.96194947,0.01554881,0.0026602645,0.0034492172,0.015254747,0.0011374387],"domain_scores_gemma":[0.8314699,0.058136884,0.0061206794,0.035534594,0.058013044,0.010724879],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05554742,0.0010386048,0.0010135232,0.0050752433,0.0030134604,0.012932111,0.0059090266,0.0068437327,0.0045121973],"category_scores_gemma":[0.08901803,0.00096243486,0.0008303528,0.003026246,0.012941385,0.013966969,0.0069312397,0.014195956,0.002497121],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000090862086,0.0010479428,0.005745389,0.0007070448,0.00006568574,0.00020404917,0.0038117187,0.0060147946,0.005351363,0.57903713,0.018872973,0.379051],"study_design_scores_gemma":[0.00015609058,0.00082336954,0.009265673,0.002939301,0.000064076245,0.0014083036,0.0036128066,0.03206409,0.010811883,0.58198726,0.35667974,0.00018739213],"about_ca_topic_score_codex":0.0023456463,"about_ca_topic_score_gemma":0.0023549867,"teacher_disagreement_score":0.05554742,"about_ca_system_score_codex":0.0043500704,"about_ca_system_score_gemma":0.015996315,"threshold_uncertainty_score":0.29376632},"labels":[],"label_agreement":null},{"id":"W2030063053","doi":"10.1142/s0129054102001497","title":"CONSTRUCTING RED-BLACK TREE SHAPES","year":2002,"lang":"en","type":"article","venue":"International Journal of Foundations of Computer Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Tree (set theory); Correctness; Binary tree; Mathematics; Combinatorics; Sequence (biology); Optimal binary search tree; Random binary tree; K-ary tree; Range tree; Algorithm; Ternary search tree; Weight-balanced tree; Interval tree; Discrete mathematics; Binary search tree; Tree structure; Chemistry","score_opus":0.03742748328499703,"score_gpt":0.30269494474112313,"score_spread":0.2652674614561261,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2030063053","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048898246,0.00019929734,0.93654996,0.00029827963,0.000071848044,0.00016528636,0.00029528575,0.0046043084,0.008917418],"genre_scores_gemma":[0.25749105,0.00020535554,0.7350606,0.00026939777,0.000018311204,0.0002087516,0.0006403474,0.001307315,0.004798952],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99831414,0.00033814227,0.0001260474,0.0003038952,0.00070493796,0.00021293788],"domain_scores_gemma":[0.99581665,0.0012432212,0.0003990954,0.001434622,0.00088696333,0.00021941637],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013105813,0.0006598573,0.0007615239,0.0010599716,0.0010824575,0.0011993644,0.0018242578,0.0011874744,0.0031459974],"category_scores_gemma":[0.008398391,0.00086246023,0.0009147714,0.0010954207,0.0016508288,0.0027335135,0.002759838,0.0014022306,0.001685022],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00065486354,0.00021070198,0.004758751,0.0005043341,0.00007480291,0.00090442476,0.001980731,0.053937614,0.08331689,0.46485373,0.011993925,0.37680927],"study_design_scores_gemma":[0.00010586861,0.00022736427,0.0014345115,0.00020805722,0.00009419372,0.0010634761,0.00045667792,0.37841833,0.1566922,0.35913128,0.102047004,0.00012107328],"about_ca_topic_score_codex":0.00093867705,"about_ca_topic_score_gemma":0.0014171482,"teacher_disagreement_score":0.0031459974,"about_ca_system_score_codex":0.0007133414,"about_ca_system_score_gemma":0.0009537748,"threshold_uncertainty_score":0.010524392},"labels":[],"label_agreement":null},{"id":"W2030553922","doi":"10.1093/comjnl/bxp073","title":"Fault Coverage-Driven Incremental Test Generation","year":2009,"lang":"en","type":"article","venue":"The Computer Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Test suite; Fault coverage; Computer science; Generalization; Completeness (order theory); Test case; Finite-state machine; Fault (geology); Algorithm; Automatic test pattern generation; Test (biology); Code coverage; Programming language; Mathematics; Software; Machine learning; Engineering","score_opus":0.026847956045254667,"score_gpt":0.2612589342968718,"score_spread":0.23441097825161714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2030553922","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037632465,0.0001040596,0.95918995,0.00010383566,0.000013906445,0.00014397151,0.00007512257,0.0017972833,0.0009394312],"genre_scores_gemma":[0.48363125,0.00010706782,0.51376414,0.00015381433,0.000025531099,0.00033801125,0.0007469504,0.00030475415,0.00092843414],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9981159,0.00075538387,0.0000929072,0.00021939487,0.0006462151,0.0001702218],"domain_scores_gemma":[0.99015194,0.007356594,0.00037453065,0.0011251365,0.0008762275,0.000115646144],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015170036,0.0007737144,0.00073938764,0.0010924756,0.00029943502,0.00048126327,0.0017082164,0.00080529397,0.0017858071],"category_scores_gemma":[0.012396972,0.0003147856,0.0008445474,0.0006565127,0.0007639279,0.0011627598,0.0012565539,0.0007233281,0.0003253499],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044277194,0.0002809675,0.00393035,0.00041970008,0.000101986545,0.00082691276,0.00027390823,0.44438046,0.031052465,0.029485755,0.0032091597,0.4855955],"study_design_scores_gemma":[0.000059710605,0.0001661758,0.00052661117,0.000021194548,0.000038372586,0.00027834173,0.000022122103,0.9648488,0.017644877,0.014573401,0.0018050766,0.000015292206],"about_ca_topic_score_codex":0.0009810092,"about_ca_topic_score_gemma":0.0014654009,"teacher_disagreement_score":0.0017858071,"about_ca_system_score_codex":0.00052684627,"about_ca_system_score_gemma":0.0009042434,"threshold_uncertainty_score":0.008022726},"labels":[],"label_agreement":null},{"id":"W2032695995","doi":"10.1145/1822090.1822100","title":"Enbug","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Debugger; Exploit; Reuse; Software engineering; Robustness (evolution); Code reuse; Software; Software security assurance; Software bug; Software construction; Software development; Programming language; Computer security; Debugging; Engineering; Information security","score_opus":0.007335514611735122,"score_gpt":0.23704308351393724,"score_spread":0.22970756890220212,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032695995","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029175987,0.0016385044,0.51498944,0.0010878369,0.0009077264,0.0005705054,0.0047957697,0.34809995,0.098734215],"genre_scores_gemma":[0.22383721,0.0013828502,0.5450353,0.0024252012,0.0001395024,0.0008406241,0.021579562,0.07736814,0.12739155],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985821,0.0003301201,0.00010817879,0.0002716188,0.00054254354,0.00016537729],"domain_scores_gemma":[0.9972395,0.0008660547,0.000114828836,0.0009501464,0.00068024785,0.00014919537],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001982998,0.000856926,0.0005991968,0.0014143054,0.0007186565,0.0016418416,0.0018305863,0.0011453922,0.038104173],"category_scores_gemma":[0.0076725734,0.0005603467,0.000472031,0.00077297783,0.0006330446,0.0040151435,0.002348109,0.0014623856,0.018508608],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001200809,0.0004569116,0.006549905,0.0009761908,0.00007914388,0.00072844006,0.0011554336,0.005566577,0.031094946,0.047720462,0.20582692,0.6986443],"study_design_scores_gemma":[0.0001934883,0.00028746403,0.0022767824,0.00025373136,0.000053729967,0.0011971936,0.00016531917,0.026624277,0.046256296,0.014897333,0.9076767,0.00011773031],"about_ca_topic_score_codex":0.0024935862,"about_ca_topic_score_gemma":0.0027036695,"teacher_disagreement_score":0.9618958,"about_ca_system_score_codex":0.00043954072,"about_ca_system_score_gemma":0.0009962698,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2032754744","doi":"10.1109/tse.2014.2371458","title":"Guided Mutation Testing for JavaScript Web Applications","year":2015,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Mutation testing; Test suite; Mutation; JavaScript; Programming language; Metric (unit); Web application; Process (computing); Set (abstract data type); Focus (optics); Data mining; Test case; Theoretical computer science; Machine learning; Operating system; Regression analysis","score_opus":0.057086510097033846,"score_gpt":0.2734156176940332,"score_spread":0.21632910759699936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032754744","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6006421,0.000985648,0.38003653,0.0004055768,0.000033013184,0.00031357692,0.00026956876,0.014532301,0.0027818216],"genre_scores_gemma":[0.86092645,0.00018094602,0.13724296,0.000104299324,0.000011804711,0.00011397134,0.00036668626,0.00034656504,0.0007063491],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955011,0.0018758,0.0002812671,0.00041089975,0.0017500209,0.00018089739],"domain_scores_gemma":[0.9876545,0.008244569,0.0015348988,0.001088125,0.0012173029,0.00026057084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025616686,0.00087337056,0.00059683487,0.0018736968,0.00052316394,0.0007575676,0.0013607338,0.00094446517,0.0007162849],"category_scores_gemma":[0.019562904,0.0003424173,0.0005994033,0.00081594154,0.00081962,0.0011771483,0.0008009861,0.00064043055,0.00027123245],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077219855,0.0008483882,0.04184237,0.00050258054,0.00016882634,0.0017401174,0.0006008318,0.2570594,0.14811869,0.007261727,0.0037242603,0.5373605],"study_design_scores_gemma":[0.00010111049,0.00040314195,0.008381547,0.0000688729,0.000070863985,0.0010189987,0.00009174015,0.9105837,0.068339474,0.008333697,0.002563318,0.000043573804],"about_ca_topic_score_codex":0.0028998125,"about_ca_topic_score_gemma":0.003925254,"teacher_disagreement_score":0.0028998125,"about_ca_system_score_codex":0.0007230003,"about_ca_system_score_gemma":0.0012019319,"threshold_uncertainty_score":0.013547599},"labels":[],"label_agreement":null},{"id":"W2034691511","doi":"10.1093/logcom/exn077","title":"Collaborative Runtime Verification with Tracematches","year":2008,"lang":"en","type":"article","venue":"Journal of Logic and Computation","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; McGill University","funders":"","keywords":"Runtime verification; Computer science; Software deployment; Overhead (engineering); Benchmark (surveying); Instrumentation (computer programming); Distributed computing; Runtime system; Static analysis; Embedded system; Formal verification; Operating system; Programming language","score_opus":0.024144604480013293,"score_gpt":0.26188223756999995,"score_spread":0.23773763308998666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034691511","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0143064335,0.00004265395,0.97458005,0.00015337706,0.00002275758,0.00010638636,0.00007141938,0.009906828,0.00081013166],"genre_scores_gemma":[0.47056434,0.00009719883,0.52302295,0.00028394323,0.00005518709,0.00044713673,0.00046562022,0.0027219483,0.0023416174],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.97232765,0.013132106,0.002120666,0.004336921,0.0067017917,0.0013807513],"domain_scores_gemma":[0.90544087,0.045042824,0.0059818756,0.03863712,0.00398006,0.0009173145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019378535,0.0015826577,0.0014365593,0.0014660074,0.0010357477,0.0036745775,0.004616581,0.0019441452,0.0041294945],"category_scores_gemma":[0.06793906,0.0015390929,0.002279887,0.0009958963,0.0035757788,0.008705227,0.0064889453,0.0030102262,0.0010194844],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020424265,0.000946586,0.013346221,0.0007600271,0.00054518593,0.0012181589,0.0018963992,0.318896,0.051123228,0.16806549,0.0069516012,0.4342086],"study_design_scores_gemma":[0.0003414405,0.00036255174,0.000748358,0.00009840174,0.00012035929,0.00037665386,0.00018351215,0.78367484,0.08886275,0.115025505,0.010091442,0.00011421086],"about_ca_topic_score_codex":0.0029575233,"about_ca_topic_score_gemma":0.003266347,"teacher_disagreement_score":0.019378535,"about_ca_system_score_codex":0.0018857655,"about_ca_system_score_gemma":0.0037138967,"threshold_uncertainty_score":0.1024847},"labels":[],"label_agreement":null},{"id":"W2034869573","doi":"10.1023/b:ause.0000008668.12782.6c","title":"Granularity-Driven Dynamic Predicate Slicing Algorithms for Message Passing Systems","year":2003,"lang":"en","type":"article","venue":"Automated Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Program slicing; Computer science; Predicate (mathematical logic); Slicing; Correctness; Granularity; Algorithm; Computation; Theoretical computer science; Programming language","score_opus":0.014441274826881756,"score_gpt":0.2554814375190623,"score_spread":0.24104016269218057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034869573","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01973156,0.00013882264,0.9755358,0.000113248534,0.00002300156,0.000035366484,0.000047410056,0.0034586715,0.0009160779],"genre_scores_gemma":[0.37354743,0.00015645078,0.6242109,0.00008721186,0.00003796776,0.0000816428,0.00024624655,0.000691174,0.0009410033],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975776,0.0007326152,0.00019336346,0.00030507037,0.00081267575,0.00037868787],"domain_scores_gemma":[0.99128515,0.004634518,0.0006108501,0.002482733,0.00074666267,0.00024002312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039852746,0.00074129354,0.0010545815,0.0013701447,0.0010964548,0.0016676418,0.002263019,0.0009274883,0.0031841798],"category_scores_gemma":[0.00896103,0.0008292561,0.000879882,0.0013609339,0.0018539478,0.0040176995,0.002519095,0.002205963,0.0004543618],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011466277,0.0001819935,0.004143053,0.00033389492,0.000105134364,0.00013000908,0.0009522913,0.27524206,0.027105873,0.24180208,0.0064781164,0.44237885],"study_design_scores_gemma":[0.00009883931,0.00007510376,0.00034317764,0.000032214044,0.00004684158,0.000039482537,0.00006284838,0.85935444,0.0154306,0.1216386,0.0028519796,0.000026003625],"about_ca_topic_score_codex":0.0043370086,"about_ca_topic_score_gemma":0.005456187,"teacher_disagreement_score":0.0043370086,"about_ca_system_score_codex":0.001605943,"about_ca_system_score_gemma":0.0018158816,"threshold_uncertainty_score":0.021076381},"labels":[],"label_agreement":null},{"id":"W2034873526","doi":"10.1007/s00446-008-0062-4","title":"Checking sequences for distributed test architectures","year":2008,"lang":"en","type":"article","venue":"Distributed Computing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Observability; Controllability; Computer science; Sequence (biology); Construct (python library); Sequence diagram; Distributed computing; Test (biology); Model checking; Digraph; Theoretical computer science; Programming language; Unified Modeling Language; Mathematics; Software","score_opus":0.033862512938887826,"score_gpt":0.2757959318795185,"score_spread":0.2419334189406307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034873526","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17181586,0.0002437488,0.81271666,0.00059731735,0.00009616583,0.0003240345,0.00046025653,0.009521416,0.004224446],"genre_scores_gemma":[0.72278017,0.00010045502,0.2713146,0.00024820757,0.000047076886,0.00026304898,0.0010828768,0.00086130685,0.0033023644],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99213785,0.0031734887,0.00070535863,0.0010678944,0.0021293,0.0007861726],"domain_scores_gemma":[0.92410505,0.056965422,0.0039744787,0.008504717,0.0052740905,0.0011761443],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062532392,0.0013497671,0.0011091955,0.0023965882,0.0012371099,0.0020923573,0.0029447468,0.0021624917,0.0056965225],"category_scores_gemma":[0.037133306,0.0009888442,0.0012771063,0.0013741952,0.0024051263,0.004468919,0.0025817698,0.0019608177,0.0008394103],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0047483565,0.0013484498,0.030144867,0.00121608,0.00031585418,0.0019563246,0.0019883548,0.21909514,0.040730108,0.2698539,0.0133208055,0.41528174],"study_design_scores_gemma":[0.00047507425,0.00055048906,0.0017978864,0.00021352689,0.00020604614,0.00044891704,0.00024484866,0.62176573,0.048810463,0.32015425,0.0052391733,0.0000935432],"about_ca_topic_score_codex":0.0035360642,"about_ca_topic_score_gemma":0.0063915993,"teacher_disagreement_score":0.0062532392,"about_ca_system_score_codex":0.0016808121,"about_ca_system_score_gemma":0.0031549747,"threshold_uncertainty_score":0.033070683},"labels":[],"label_agreement":null},{"id":"W2036316394","doi":"10.1145/2814270.2814297","title":"SATCheck: SAT-directed stateless model checking for SC and TSO","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Science Foundation","keywords":"Computer science; Model checking; Scalability; Stateless protocol; Parallel computing; Memory model; Programming language; Thread (computing); Concurrency; Shared memory; Operating system","score_opus":0.09452424806707642,"score_gpt":0.3089662886004712,"score_spread":0.21444204053339477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036316394","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021292357,0.00013110264,0.94845325,0.00029799138,0.00006993815,0.00016243281,0.0009996461,0.026108181,0.0024851295],"genre_scores_gemma":[0.3857632,0.00014715518,0.60550916,0.00042672962,0.000046851772,0.0004318825,0.0025960915,0.0023711747,0.0027076283],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99622667,0.0013184868,0.00028721945,0.000739451,0.0011072076,0.00032102875],"domain_scores_gemma":[0.9855223,0.009962513,0.00087054796,0.002259363,0.0012232331,0.0001621002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028087331,0.001569833,0.0008910456,0.0014602017,0.00091447425,0.0016688927,0.0030325023,0.0011323532,0.0070861075],"category_scores_gemma":[0.015596137,0.0010641112,0.0030467152,0.0011331576,0.0026237597,0.0032187812,0.0022404438,0.0023463536,0.001024282],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001350453,0.00041124396,0.011699608,0.0014812635,0.0005042904,0.00062092906,0.00052019407,0.6377021,0.023825468,0.14098753,0.01742975,0.16346715],"study_design_scores_gemma":[0.00016222353,0.000108703054,0.00027509153,0.000057214893,0.00006893691,0.000082306906,0.0000504548,0.9359555,0.016983785,0.042772334,0.003457478,0.000025910505],"about_ca_topic_score_codex":0.013119555,"about_ca_topic_score_gemma":0.027214559,"teacher_disagreement_score":0.013119555,"about_ca_system_score_codex":0.0019046295,"about_ca_system_score_gemma":0.005279253,"threshold_uncertainty_score":0.02608639},"labels":[],"label_agreement":null},{"id":"W2036633739","doi":"10.1109/trustcom.2011.151","title":"Tool Support for Automated Traceability of Test/Code Artifacts in Embedded Software Systems","year":2011,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Traceability; Computer science; Software engineering; Embedded system; Software development; Software construction; Verification and validation; Static program analysis; Development testing; Unit testing; Software; Source code; Software system; Embedded software; Software framework; Operating system; Engineering","score_opus":0.05771103409542019,"score_gpt":0.28947700816024985,"score_spread":0.23176597406482966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036633739","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016716199,0.00014755805,0.95357645,0.0000976482,0.000012389998,0.00011478501,0.00014721551,0.02827087,0.00091685954],"genre_scores_gemma":[0.25878567,0.00032562212,0.7348931,0.000084248495,0.00002414667,0.00025327597,0.0011717394,0.003015185,0.001447048],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9958954,0.0014287257,0.00049841276,0.0004455612,0.0015554676,0.00017644945],"domain_scores_gemma":[0.9693582,0.021211693,0.0016900061,0.005796073,0.0016478755,0.00029619643],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004830937,0.0013119596,0.00090124493,0.0035766913,0.00048141403,0.0023786803,0.0022444949,0.0011715904,0.002765082],"category_scores_gemma":[0.027357059,0.000806954,0.0008968,0.0016281666,0.0008135065,0.0024431986,0.0015149502,0.0017973781,0.00097334577],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037032762,0.00043502625,0.005942113,0.0009650121,0.0001506012,0.0017101464,0.0015768723,0.07251501,0.071686916,0.021840956,0.0051968023,0.81761014],"study_design_scores_gemma":[0.00021842301,0.0004122902,0.003587094,0.00066903955,0.00014817699,0.001699872,0.00020261294,0.7912766,0.13799693,0.033689268,0.029964434,0.00013530786],"about_ca_topic_score_codex":0.0014075434,"about_ca_topic_score_gemma":0.0012646334,"teacher_disagreement_score":0.004830937,"about_ca_system_score_codex":0.00043251933,"about_ca_system_score_gemma":0.0011654095,"threshold_uncertainty_score":0.025548756},"labels":[],"label_agreement":null},{"id":"W2036896594","doi":"10.1016/j.infsof.2013.02.006","title":"A systematic mapping study of web application testing","year":2013,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":108,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of Calgary","funders":"","keywords":"Bibliometrics; Computer science; Data science; Field (mathematics); Schema (genetic algorithms); Information retrieval; Systematic review; World Wide Web; MEDLINE","score_opus":0.012898108410190436,"score_gpt":0.22467908839648615,"score_spread":0.21178097998629572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036896594","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89466536,0.0067086294,0.08040228,0.00055325514,0.000073594274,0.003449189,0.001448755,0.00045242306,0.012246498],"genre_scores_gemma":[0.9499515,0.0014309052,0.04537695,0.00017735263,0.0000112214575,0.0010215734,0.0006863892,0.00007565082,0.001268395],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.9773312,0.015492547,0.0022869299,0.0013547506,0.0030207231,0.0005138636],"domain_scores_gemma":[0.7169875,0.208155,0.015801186,0.025045266,0.0326645,0.0013465756],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.014886764,0.0005885662,0.0007608939,0.0138231125,0.0021201638,0.0018034779,0.0014781422,0.0008354512,0.0025630668],"category_scores_gemma":[0.14815946,0.00068923604,0.0008829731,0.009945185,0.0015802982,0.003944129,0.0032142606,0.00093690923,0.00040278555],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006470575,0.0021425262,0.25732145,0.009764352,0.0006498025,0.0013013158,0.057742164,0.0022432741,0.0063843345,0.013356048,0.0026270086,0.6458206],"study_design_scores_gemma":[0.00073994213,0.0077265757,0.6413473,0.022844506,0.0055136625,0.008370626,0.108376265,0.036783732,0.043939173,0.044790037,0.07913979,0.0004283645],"about_ca_topic_score_codex":0.01283384,"about_ca_topic_score_gemma":0.023853134,"teacher_disagreement_score":0.98511326,"about_ca_system_score_codex":0.0025167838,"about_ca_system_score_gemma":0.012990015,"threshold_uncertainty_score":0.07872969},"labels":[],"label_agreement":null},{"id":"W2040768172","doi":"10.1145/2304510.2304512","title":"BloomUnit","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Air Force Office of Scientific Research; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Microsoft Research","keywords":"Computer science; Programming language; Semantics (computer science); Bloom filter; Software; Test (biology); Operating system","score_opus":0.02562895817678755,"score_gpt":0.2599044611156117,"score_spread":0.23427550293882415,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2040768172","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059378888,0.00027172762,0.95515656,0.00024057555,0.00005958332,0.00026768816,0.0005317671,0.028575078,0.008959243],"genre_scores_gemma":[0.15846835,0.00047195517,0.81735945,0.0006394127,0.000060600963,0.0006302516,0.0028121122,0.0046689752,0.014888939],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99688536,0.000880085,0.00024294097,0.0005007061,0.0010653702,0.00042555106],"domain_scores_gemma":[0.9953107,0.0020948355,0.00031884885,0.001130799,0.0009239414,0.00022087363],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035981936,0.001270012,0.0010175966,0.0018842482,0.0012311835,0.002793278,0.003664231,0.0012942269,0.013855606],"category_scores_gemma":[0.009639522,0.0011489721,0.001439561,0.0010434247,0.0018571685,0.0051126573,0.0034663756,0.0019576112,0.0035405045],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007881508,0.00034577498,0.006656974,0.00090955343,0.000118388525,0.0007282471,0.00075591,0.05608055,0.012955661,0.42265162,0.04546809,0.4525411],"study_design_scores_gemma":[0.0001720874,0.00044515746,0.0010019082,0.00042465414,0.00011614293,0.0011884691,0.00022560956,0.44935337,0.04919636,0.24018669,0.2575443,0.00014517715],"about_ca_topic_score_codex":0.0087985955,"about_ca_topic_score_gemma":0.010577118,"teacher_disagreement_score":0.013855606,"about_ca_system_score_codex":0.0024289917,"about_ca_system_score_gemma":0.0041683675,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2040973615","doi":"10.5555/777092.777204","title":"Optimal depth-first strategies for and-or trees","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of Alberta","funders":"","keywords":"Probabilistic logic; Computer science; Node (physics); Set (abstract data type); Tree (set theory); Task (project management); Test (biology); Sequence (biology); Focus (optics); Algorithm; Mathematical optimization; Mathematics; Theoretical computer science; Artificial intelligence; Combinatorics","score_opus":0.061381381103267206,"score_gpt":0.28324691919071004,"score_spread":0.22186553808744283,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2040973615","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13489798,0.0013088385,0.84724915,0.0011450042,0.00005338739,0.0005586993,0.0005419935,0.0013332607,0.012911694],"genre_scores_gemma":[0.3921506,0.0007023926,0.59831685,0.000552262,0.000036142403,0.00044165817,0.00074619113,0.0004328636,0.0066210106],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99791354,0.000676563,0.00014423924,0.0003881423,0.00046722952,0.0004102662],"domain_scores_gemma":[0.99387276,0.004784086,0.00039383618,0.00037030297,0.00032896645,0.00025001066],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022009748,0.0012957206,0.0016030524,0.0015079877,0.0010782497,0.0020465867,0.0017576707,0.0017752742,0.0048783678],"category_scores_gemma":[0.011812659,0.0010236669,0.0013773083,0.0010069147,0.0014144516,0.003909227,0.0015856307,0.0016906607,0.00073298236],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006276859,0.0004605599,0.0026363658,0.00073058583,0.000156166,0.0003987608,0.001164556,0.30297753,0.015857633,0.37740296,0.010453267,0.28713402],"study_design_scores_gemma":[0.00016757566,0.00016599242,0.00036768246,0.00007850968,0.000082293685,0.00021401777,0.00017957528,0.5898223,0.006465718,0.39703518,0.0053688046,0.000052265717],"about_ca_topic_score_codex":0.0036587198,"about_ca_topic_score_gemma":0.0058501796,"teacher_disagreement_score":0.0048783678,"about_ca_system_score_codex":0.002933727,"about_ca_system_score_gemma":0.0030165005,"threshold_uncertainty_score":0.021285772},"labels":[],"label_agreement":null},{"id":"W2042380289","doi":"10.1109/ccece.2010.5575202","title":"Specification-based test oracles with JUnit","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"Memorial University of Newfoundland","keywords":"Computer science; Oracle; Test Management Approach; Test case; Test harness; Software engineering; Programming language; Process (computing); Test (biology); Unit testing; Software; Software development; Software construction","score_opus":0.01030708878419913,"score_gpt":0.2239426817707521,"score_spread":0.21363559298655296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2042380289","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008196075,0.0001896563,0.8947444,0.00016229623,0.00011343012,0.00017931416,0.0005572507,0.088736825,0.0071208063],"genre_scores_gemma":[0.3022062,0.00042174628,0.6516496,0.0005168666,0.00015557332,0.00077525375,0.0061852764,0.028558522,0.0095309755],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9918936,0.0025681464,0.0010363611,0.0010603595,0.0028675708,0.0005738739],"domain_scores_gemma":[0.97892237,0.0092247715,0.0015213632,0.007299664,0.0025798015,0.0004520011],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010550608,0.0019279574,0.0009528112,0.002071422,0.0007004604,0.004205166,0.0040332265,0.0014763317,0.011206285],"category_scores_gemma":[0.02631654,0.0014156395,0.0015761189,0.0013833632,0.0022834712,0.0050324877,0.0037008426,0.0031641223,0.0037176108],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0042384225,0.00086929335,0.014014416,0.0017187193,0.00046666083,0.002821877,0.002162789,0.102766074,0.03000904,0.22596696,0.057688847,0.55727684],"study_design_scores_gemma":[0.0006765827,0.00072717125,0.0019526888,0.00042499581,0.0002575779,0.001819943,0.00021782133,0.63177294,0.11370274,0.11529356,0.13283332,0.00032064875],"about_ca_topic_score_codex":0.0017313344,"about_ca_topic_score_gemma":0.0018854882,"teacher_disagreement_score":0.011206285,"about_ca_system_score_codex":0.0011252922,"about_ca_system_score_gemma":0.0013717893,"threshold_uncertainty_score":0.055797577},"labels":[],"label_agreement":null},{"id":"W2043042437","doi":"10.1109/issre.2013.6698880","title":"Feedback-directed exploration of web applications to derive test models","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Test (biology); Geology","score_opus":0.04747123082249917,"score_gpt":0.2710008017299474,"score_spread":0.22352957090744824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2043042437","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10839354,0.00015831624,0.88384247,0.00012658355,0.000008433397,0.00030617116,0.00027794033,0.004778282,0.0021082817],"genre_scores_gemma":[0.58953524,0.000120942146,0.40759534,0.000087453554,0.000008337983,0.0005027795,0.00093628577,0.00040283013,0.00081089884],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983814,0.00063403585,0.00009975502,0.00018128104,0.00058477005,0.00011880234],"domain_scores_gemma":[0.9867845,0.010661465,0.00051466876,0.0008809291,0.0010314935,0.00012685094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001532728,0.0010821673,0.00060736464,0.0019334382,0.00037159704,0.00075327564,0.00097307266,0.000876684,0.0015728051],"category_scores_gemma":[0.01658186,0.00056911324,0.0011497242,0.00073160796,0.0006106639,0.001385703,0.0014268775,0.0007992138,0.00030612515],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020984869,0.00037234704,0.010741876,0.00042271748,0.0001067307,0.00050937384,0.0007061991,0.7556896,0.023657281,0.009640054,0.0015992455,0.19634473],"study_design_scores_gemma":[0.000011192857,0.00006340326,0.00043200134,0.000020117383,0.000010874695,0.0000722902,0.00003789573,0.987927,0.0064102337,0.004420834,0.00058663636,0.000007400024],"about_ca_topic_score_codex":0.0031371345,"about_ca_topic_score_gemma":0.0065609477,"teacher_disagreement_score":0.0031371345,"about_ca_system_score_codex":0.00074302993,"about_ca_system_score_gemma":0.0013125676,"threshold_uncertainty_score":0.008105993},"labels":[],"label_agreement":null},{"id":"W2043839920","doi":"10.1016/j.jss.2014.01.010","title":"Web application testing: A systematic literature review","year":2014,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":108,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Web testing; Dependability; Systematic review; Field (mathematics); Web application; World Wide Web; Empirical research; Data science; Set (abstract data type); Web engineering; Software engineering; Information retrieval; Web application security; The Internet; Web development","score_opus":0.016224326302853694,"score_gpt":0.2516204591044873,"score_spread":0.23539613280163363,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2043839920","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036591985,0.992063,0.0006897793,0.000659054,0.00016969643,0.0015911706,0.00072226534,0.000020234942,0.00042561183],"genre_scores_gemma":[0.038740765,0.9529697,0.003952641,0.0014322312,0.00010700473,0.0020294206,0.0005408912,0.000021528558,0.00020584083],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.9722,0.008684373,0.012288679,0.0016880136,0.0045206924,0.0006182463],"domain_scores_gemma":[0.8744955,0.089276776,0.0206444,0.0023385186,0.01182777,0.0014169285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.031157574,0.0019724474,0.009482507,0.02916416,0.0013514548,0.0037117344,0.0028427516,0.0029651104,0.0040637003],"category_scores_gemma":[0.10223705,0.0017482558,0.0073131244,0.019585714,0.0018326174,0.00589163,0.0036405397,0.0017113204,0.00043506204],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002664805,0.000067230736,0.00205085,0.92086715,0.0068227854,0.00027016504,0.0007587859,0.00011456898,0.00033787993,0.0002194546,0.0025303506,0.06569418],"study_design_scores_gemma":[0.0003472883,0.00025035674,0.0044636484,0.9331358,0.04676914,0.0005912998,0.0010700739,0.000106614396,0.00024844584,0.00032777563,0.012628504,0.00006104765],"about_ca_topic_score_codex":0.008633388,"about_ca_topic_score_gemma":0.042756006,"teacher_disagreement_score":0.031157574,"about_ca_system_score_codex":0.0061128684,"about_ca_system_score_gemma":0.028302735,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2044138969","doi":"10.1016/s0140-3664(03)00116-6","title":"Context independent unique state identification sequences for testing communication protocols modelled as extended finite state machines","year":2003,"lang":"en","type":"article","venue":"Computer Communications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; Nortel (Canada)","funders":"","keywords":"Extended finite-state machine; Computer science; Finite-state machine; Executable; Observability; Sequence (biology); Identification (biology); Conformance testing; Context (archaeology); State (computer science); Theoretical computer science; Control flow; Algorithm; Mathematics; Programming language","score_opus":0.09491792874771247,"score_gpt":0.3503298570391113,"score_spread":0.2554119282913988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2044138969","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14649078,0.0002882292,0.84826374,0.00017399745,0.000054169966,0.00020619402,0.00016938055,0.0031177944,0.0012357532],"genre_scores_gemma":[0.8030507,0.0001030736,0.1950382,0.00013692697,0.000030837655,0.0003736655,0.00032949916,0.00025044894,0.0006866625],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9874669,0.0055910065,0.00096815266,0.0017904687,0.0031427883,0.0010407089],"domain_scores_gemma":[0.9284917,0.05482225,0.004777742,0.007881508,0.0029366654,0.0010900687],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077997902,0.0017016417,0.0017124059,0.002290123,0.0008506042,0.0016293057,0.0028286993,0.0022285944,0.0017722212],"category_scores_gemma":[0.03747699,0.0010781181,0.0014318874,0.0010897804,0.0033765868,0.0057746535,0.0032791158,0.0034342324,0.00026662377],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004893072,0.0011053708,0.010764751,0.0011692446,0.0003328228,0.0016004161,0.0015972081,0.45197842,0.037815977,0.32085985,0.0019149414,0.16596797],"study_design_scores_gemma":[0.00016869236,0.00054568565,0.0006998522,0.0001587939,0.00013824346,0.00027661677,0.000110869536,0.8143495,0.027250715,0.15544395,0.00078618276,0.0000710112],"about_ca_topic_score_codex":0.001176895,"about_ca_topic_score_gemma":0.0016827504,"teacher_disagreement_score":0.0077997902,"about_ca_system_score_codex":0.0014744259,"about_ca_system_score_gemma":0.0025831994,"threshold_uncertainty_score":0.041249692},"labels":[],"label_agreement":null},{"id":"W2044863546","doi":"10.1007/s10009-013-0277-y","title":"Generating effective tests for concurrent programs via AI automated planning techniques","year":2013,"lang":"en","type":"article","venue":"International Journal on Software Tools for Technology Transfer","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Concurrency; Atomicity; Correctness; Interleaving; Programming language; Debugging; Set (abstract data type); Theory of computation","score_opus":0.029292245096745266,"score_gpt":0.33535090047419397,"score_spread":0.3060586553774487,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2044863546","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04547626,0.00013896232,0.9447534,0.00019579811,0.0000391885,0.00026612368,0.00020832288,0.00508102,0.0038409936],"genre_scores_gemma":[0.44600582,0.00010369925,0.5518357,0.00009564134,0.000024631929,0.00028060545,0.00034892772,0.00044906273,0.0008559314],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99726593,0.000769824,0.00015633629,0.0003600743,0.0011699187,0.00027787237],"domain_scores_gemma":[0.9805736,0.016200777,0.0007819539,0.0010172025,0.0012127182,0.000213801],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015886282,0.0015722997,0.00081355573,0.002077835,0.00086322735,0.0013627075,0.002050118,0.0010380103,0.004909793],"category_scores_gemma":[0.016197748,0.0008104877,0.0013142808,0.0013589206,0.0019349158,0.001908624,0.0016767628,0.0015240649,0.0005015473],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072092435,0.0005395392,0.0047800434,0.0010064569,0.00016430474,0.0014451505,0.0006772916,0.5215262,0.042417537,0.055181332,0.003915852,0.36762542],"study_design_scores_gemma":[0.00016250848,0.0002503221,0.00051349396,0.00005326083,0.000107944,0.00015009596,0.00015701582,0.9175482,0.029375667,0.049796514,0.0018515997,0.000033522032],"about_ca_topic_score_codex":0.005867213,"about_ca_topic_score_gemma":0.011646083,"teacher_disagreement_score":0.005867213,"about_ca_system_score_codex":0.0011707966,"about_ca_system_score_gemma":0.0026184567,"threshold_uncertainty_score":0.016424835},"labels":[],"label_agreement":null},{"id":"W2045837563","doi":"10.1007/s10664-012-9219-7","title":"Static test case prioritization using topic models","year":2012,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":156,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Test suite; Computer science; Code coverage; Test case; Source code; Black box; Test Management Approach; Programming language; White-box testing; Test (biology); Test data; Operating system; Software; Software development; Artificial intelligence; Machine learning","score_opus":0.06432137932201937,"score_gpt":0.3036697114581389,"score_spread":0.23934833213611956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2045837563","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15312707,0.00089590263,0.82645845,0.00080564176,0.00013584176,0.00088188215,0.0015836222,0.009680969,0.0064306594],"genre_scores_gemma":[0.78265876,0.00029060998,0.20888586,0.00012865239,0.00013079308,0.00074050645,0.0029775435,0.0011024416,0.0030848272],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98874354,0.0058841496,0.0008405673,0.001650548,0.002206416,0.0006749231],"domain_scores_gemma":[0.9297371,0.056032773,0.0021194858,0.003859625,0.007100894,0.0011501584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010082978,0.0017337598,0.0017287617,0.011113924,0.0010781361,0.003485714,0.0024311128,0.0019223097,0.007127939],"category_scores_gemma":[0.06890455,0.0010141858,0.0025149765,0.0052286875,0.00052674615,0.0047024703,0.0020739667,0.00204661,0.001583351],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022984557,0.0011208709,0.07243875,0.0011900079,0.00088609726,0.00049733196,0.0028960344,0.10889537,0.012467312,0.022688368,0.017262397,0.75735897],"study_design_scores_gemma":[0.00022694818,0.000357278,0.010594473,0.00009439843,0.00060665875,0.00033768095,0.0006551347,0.9425092,0.007078066,0.03263905,0.004808185,0.00009302979],"about_ca_topic_score_codex":0.008395957,"about_ca_topic_score_gemma":0.0128165735,"teacher_disagreement_score":0.011113924,"about_ca_system_score_codex":0.0019911523,"about_ca_system_score_gemma":0.0034481368,"threshold_uncertainty_score":0.05332452},"labels":[],"label_agreement":null},{"id":"W2046760913","doi":"10.1109/icmla.2011.69","title":"Fault Detection through Sequential Filtering of Novelty Patterns","year":2011,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University","funders":"","keywords":"Computer science; Debugging; Concurrency; Novelty; Fault detection and isolation; Software; Domain (mathematical analysis); Preprocessor; Set (abstract data type); Data mining; Computer engineering; Parallel computing; Distributed computing; Programming language; Artificial intelligence","score_opus":0.0841736094413681,"score_gpt":0.2738196785469875,"score_spread":0.18964606910561943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2046760913","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09543736,0.00017556812,0.90238327,0.00013208164,0.000027932032,0.00012859718,0.00010831726,0.0011286573,0.00047831074],"genre_scores_gemma":[0.51499444,0.000117175005,0.48277426,0.000075770986,0.000049478844,0.0001527124,0.00045121234,0.00009533841,0.0012896659],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973368,0.00046197724,0.00030167896,0.00059862173,0.0010624003,0.00023848868],"domain_scores_gemma":[0.9814228,0.0102573335,0.002309436,0.0015836864,0.003922025,0.00050475553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028623913,0.00081497873,0.0012482265,0.003871167,0.0006119227,0.0012999144,0.0015845418,0.001101224,0.00095852924],"category_scores_gemma":[0.018352002,0.00048165818,0.0010630453,0.0017189106,0.00072981,0.0017660914,0.0012222502,0.0008084514,0.0003715223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009354107,0.0003008285,0.042400144,0.00022459125,0.00016811132,0.0003056182,0.00040893754,0.045249395,0.03641293,0.004515542,0.0013602345,0.8677184],"study_design_scores_gemma":[0.00005273142,0.00035478297,0.014582505,0.000031066917,0.000083446954,0.00064902427,0.0000927034,0.9493187,0.024547495,0.008517679,0.0017157377,0.000054096727],"about_ca_topic_score_codex":0.0035908015,"about_ca_topic_score_gemma":0.0047140247,"teacher_disagreement_score":0.003871167,"about_ca_system_score_codex":0.000681998,"about_ca_system_score_gemma":0.0012192944,"threshold_uncertainty_score":0.01513797},"labels":[],"label_agreement":null},{"id":"W2047291464","doi":"10.1109/issre.2008.57","title":"Discovering&amp;#x0A0;&amp;#x0A0;the Fault Origin from Field Traces","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Fault (geology); Computer science; Field (mathematics); Geology; Seismology; Mathematics","score_opus":0.06807463427558917,"score_gpt":0.29667765318752226,"score_spread":0.22860301891193308,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2047291464","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23042199,0.0004808958,0.75912875,0.00041440438,0.00007200883,0.00013807083,0.00088373164,0.006746861,0.0017132116],"genre_scores_gemma":[0.6876597,0.00022081689,0.30766147,0.00007679697,0.000041761876,0.00005083445,0.0013071725,0.00049306237,0.0024884462],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99921834,0.0001266394,0.00005432893,0.00019019892,0.00034916907,0.00006134759],"domain_scores_gemma":[0.99033666,0.0056550126,0.001390839,0.0011426952,0.0012368214,0.00023798071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006318471,0.0006109177,0.0005767846,0.0024245833,0.00038197372,0.0009908255,0.0010361356,0.0005944649,0.0016018471],"category_scores_gemma":[0.007778482,0.00031872737,0.0002818873,0.0009780495,0.00039162138,0.0018542258,0.0005338129,0.0008415046,0.0005643629],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008914149,0.00028207604,0.05411341,0.00058712345,0.00009121514,0.0012829466,0.00095498195,0.017004048,0.08413461,0.0083346255,0.0037114176,0.828612],"study_design_scores_gemma":[0.00010366343,0.00073885906,0.041146953,0.00021221404,0.00016215602,0.0038418954,0.00071861426,0.6474491,0.24738893,0.034408327,0.02367348,0.00015587687],"about_ca_topic_score_codex":0.0026489974,"about_ca_topic_score_gemma":0.00334701,"teacher_disagreement_score":0.0026489974,"about_ca_system_score_codex":0.00033472985,"about_ca_system_score_gemma":0.0006663688,"threshold_uncertainty_score":0.005358696},"labels":[],"label_agreement":null},{"id":"W2048753738","doi":"10.1145/2432497.2432502","title":"Model transformation testing","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Correctness; Computer science; Model transformation; Transformation (genetics); Model-based testing; Software engineering; Integration testing; Key (lock); Programming language; Software; Artificial intelligence; Test case; Machine learning","score_opus":0.09826907079709217,"score_gpt":0.28726679453888104,"score_spread":0.18899772374178886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2048753738","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033513974,0.0008414652,0.91832685,0.0015103505,0.00061679084,0.00070027204,0.00074234756,0.008748553,0.0349994],"genre_scores_gemma":[0.54563046,0.0012030583,0.4291467,0.001829653,0.00018943782,0.0011916214,0.003958986,0.0030489918,0.013801148],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.987177,0.0042235106,0.0006987889,0.0013102643,0.005927808,0.0006626356],"domain_scores_gemma":[0.97806096,0.010720655,0.00091477815,0.0056543713,0.004375189,0.00027397397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005041093,0.0016800098,0.0010353117,0.0018205618,0.0008218587,0.0020013284,0.0027277688,0.0019843935,0.009061458],"category_scores_gemma":[0.03696644,0.0004389828,0.0015356007,0.0015826591,0.0015254711,0.0037538763,0.0024457043,0.0021507794,0.003275877],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006632597,0.00094592315,0.0080539305,0.0010422355,0.00023986261,0.0016110104,0.0009772973,0.03909924,0.035955947,0.16175161,0.024666525,0.72499317],"study_design_scores_gemma":[0.00040912954,0.0014384608,0.0034417866,0.0008709423,0.00029764583,0.003079301,0.0008072394,0.41701862,0.16141233,0.24146621,0.16956379,0.00019455118],"about_ca_topic_score_codex":0.0013800964,"about_ca_topic_score_gemma":0.0011745391,"teacher_disagreement_score":0.009061458,"about_ca_system_score_codex":0.0010338756,"about_ca_system_score_gemma":0.0021982423,"threshold_uncertainty_score":0.030313611},"labels":[],"label_agreement":null},{"id":"W2049046432","doi":"10.1109/qsic.2007.4385488","title":"Alternative B-Sequences","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Sequence (biology); Finite-state machine; Computer science; Invertible matrix; Model checking; State (computer science); Algorithm; Automaton; Conformance testing; Theoretical computer science; Programming language; Mathematics","score_opus":0.02940926357054684,"score_gpt":0.3015885118565895,"score_spread":0.2721792482860427,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2049046432","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1799765,0.00040875233,0.7789973,0.0010397224,0.0003618559,0.00047611812,0.00088282675,0.002734099,0.03512286],"genre_scores_gemma":[0.5156809,0.00022572502,0.46972436,0.0007890962,0.00007017132,0.0007355543,0.0011803196,0.000442141,0.011151703],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976738,0.0006258224,0.0003055931,0.0004746206,0.0006457928,0.0002744473],"domain_scores_gemma":[0.9845785,0.007929225,0.0011272384,0.0021609368,0.0033418578,0.000862149],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021210467,0.0006950871,0.00032433195,0.0011497928,0.0009772023,0.0011851682,0.00082490843,0.0014419999,0.009953071],"category_scores_gemma":[0.011918537,0.00043596767,0.00047363958,0.00071715,0.0009879745,0.0018256614,0.0010635792,0.0016537288,0.002596917],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026553108,0.00072596496,0.012019632,0.0008335534,0.000084361265,0.0018911965,0.0014538787,0.030189412,0.13380475,0.52184165,0.0077433675,0.2867569],"study_design_scores_gemma":[0.00039725148,0.001764288,0.004079002,0.0007667618,0.00014637367,0.0026725964,0.00085489405,0.20404677,0.14004678,0.54836714,0.09665756,0.0002005824],"about_ca_topic_score_codex":0.0006441183,"about_ca_topic_score_gemma":0.0013910414,"teacher_disagreement_score":0.009953071,"about_ca_system_score_codex":0.0004830889,"about_ca_system_score_gemma":0.0011247686,"threshold_uncertainty_score":0.033296347},"labels":[],"label_agreement":null},{"id":"W2050254119","doi":"10.1109/icstw.2013.49","title":"On Adequacy of Assertions in Automated Test Suites: An Empirical Investigation","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; Türkiye Bilimsel ve Teknolojik Araştırma Kurumu","keywords":"Computer science; Oracle; Assertion; Test script; Test (biology); Context (archaeology); Programming language; Unit testing; Test case; Test suite; Source code; Code coverage; Class (philosophy); Empirical research; Keyword-driven testing; Test Management Approach; Test data; Artificial intelligence; Software; Machine learning; Statistics; Mathematics; Software development","score_opus":0.040256608223908114,"score_gpt":0.3265405027542841,"score_spread":0.28628389453037595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2050254119","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99427277,0.00021786173,0.0046112337,0.00007238711,0.000004367571,0.000072510644,0.00015366616,0.000043112956,0.0005520979],"genre_scores_gemma":[0.9976417,0.000059910828,0.0018606794,0.0000129743685,0.000007853849,0.000037447535,0.00029328186,0.000017738208,0.000068537476],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9303672,0.0328017,0.007952859,0.0061883316,0.021039804,0.0016501714],"domain_scores_gemma":[0.21862632,0.69031096,0.048592836,0.018215224,0.02240186,0.0018528177],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.036401175,0.00064455334,0.0006902132,0.0053525646,0.0008448428,0.0015946802,0.0017805379,0.0013970182,0.00096010533],"category_scores_gemma":[0.3522152,0.0004186438,0.00085587986,0.0040967665,0.0032131576,0.0031140563,0.0020207393,0.0014372598,0.00018710876],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005579002,0.0006867353,0.9387761,0.0004158614,0.00024465495,0.00077113457,0.005723447,0.008517653,0.0025382582,0.0014548319,0.00041328362,0.03990021],"study_design_scores_gemma":[0.0000901742,0.0031247993,0.877016,0.00028727378,0.00038118,0.0039613247,0.007847764,0.09222099,0.009314078,0.0027747455,0.0028826587,0.00009910803],"about_ca_topic_score_codex":0.0023911723,"about_ca_topic_score_gemma":0.0025071155,"teacher_disagreement_score":0.036401175,"about_ca_system_score_codex":0.0012119204,"about_ca_system_score_gemma":0.0009276947,"threshold_uncertainty_score":0.19251007},"labels":[],"label_agreement":null},{"id":"W2050254654","doi":"10.1007/s00165-009-0135-6","title":"Lower bounds on lengths of checking sequences","year":2009,"lang":"en","type":"article","venue":"Formal Aspects of Computing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Theory of computation; Sequence (biology); State (computer science); Model checking; Computer science; Algorithm; Upper and lower bounds; Finite-state machine; Mathematics; Theoretical computer science; Combinatorics","score_opus":0.016043546045305543,"score_gpt":0.2687929526499507,"score_spread":0.2527494066046451,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2050254654","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1322946,0.0022012605,0.84722555,0.0015972133,0.00025582843,0.0002810074,0.0011424723,0.004730681,0.010271357],"genre_scores_gemma":[0.7010119,0.0009200721,0.28864142,0.0005513487,0.000451552,0.0007210064,0.0024773197,0.0018920433,0.0033333374],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9493449,0.012286887,0.0047506723,0.0066012596,0.023177098,0.0038391945],"domain_scores_gemma":[0.5551005,0.3574417,0.017688418,0.042078517,0.021496564,0.006194373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019101933,0.0019161316,0.001859848,0.00533325,0.0019333306,0.0056881797,0.0044513727,0.0030201795,0.0075318348],"category_scores_gemma":[0.17254637,0.0020757616,0.002009767,0.0033455754,0.0046773003,0.012330367,0.005366741,0.009205994,0.002178696],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0046264846,0.0008085728,0.024049275,0.0015594615,0.00035238668,0.00061527465,0.0016763471,0.35381776,0.048809916,0.28188634,0.0072416766,0.2745565],"study_design_scores_gemma":[0.00020363128,0.0007502641,0.003208052,0.00038383636,0.00017224488,0.0006002914,0.00026274758,0.69128555,0.056427717,0.23835258,0.008213779,0.0001393253],"about_ca_topic_score_codex":0.0010270086,"about_ca_topic_score_gemma":0.0012900095,"teacher_disagreement_score":0.019101933,"about_ca_system_score_codex":0.0035864026,"about_ca_system_score_gemma":0.0040917057,"threshold_uncertainty_score":0.101021886},"labels":[],"label_agreement":null},{"id":"W2050316559","doi":"10.1109/cec.2013.6557873","title":"Dynamic white-box software testing using a recursive hybrid evolutionary strategy/genetic algorithm","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Larus Technologies (Canada); University of Ottawa","funders":"","keywords":"Computer science; Programmer; Code coverage; Software; White-box testing; Test case; Software development; Programming language; Software engineering; Software construction; Machine learning","score_opus":0.023485224363464535,"score_gpt":0.2549484371156779,"score_spread":0.23146321275221335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2050316559","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037684433,0.00011844961,0.95750964,0.000115400355,0.00002114824,0.00011086054,0.00001528616,0.00062405306,0.0038006944],"genre_scores_gemma":[0.4120943,0.00011359949,0.58391094,0.00014998911,0.00001533267,0.00031429515,0.00007128923,0.00010643349,0.0032238408],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992712,0.00029285965,0.000028605633,0.000107355096,0.00021453263,0.00008548049],"domain_scores_gemma":[0.9989723,0.00073607446,0.00005950612,0.00007354671,0.0001230694,0.000035576093],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011232445,0.00082137523,0.0008268306,0.0012739465,0.0004838299,0.00073293346,0.0017080928,0.0014268326,0.0019649738],"category_scores_gemma":[0.0029545464,0.0004592744,0.0008383644,0.0007012503,0.0007876607,0.00069052435,0.0009331293,0.00060798816,0.00028417556],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000078784695,0.00015717116,0.0012903728,0.00006077759,0.00008272565,0.00021003564,0.00016046256,0.8545869,0.0105202915,0.019620063,0.0005438478,0.11268854],"study_design_scores_gemma":[0.000017305125,0.000046812285,0.00010761522,0.000006591779,0.000012536752,0.00003685361,0.000008058205,0.9967212,0.00082992145,0.001766768,0.00044007305,0.0000062509657],"about_ca_topic_score_codex":0.0042796778,"about_ca_topic_score_gemma":0.0040592123,"teacher_disagreement_score":0.0042796778,"about_ca_system_score_codex":0.0007146823,"about_ca_system_score_gemma":0.0009214104,"threshold_uncertainty_score":0.008509517},"labels":[],"label_agreement":null},{"id":"W2051066030","doi":"10.1007/s10009-013-0278-x","title":"Innovation and evolution in integrated web application testing with TTCN-3","year":2013,"lang":"en","type":"article","venue":"International Journal on Software Tools for Technology Transfer","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Integration testing; Software engineering; Vendor; Web service; Web application security; Web application; Test strategy; Unit testing; Model-based testing; Conformance testing; Web development; World Wide Web; Test case; Operating system; Software","score_opus":0.02034142140810467,"score_gpt":0.26195024814189677,"score_spread":0.2416088267337921,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2051066030","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7848582,0.00014819605,0.18849052,0.00025940154,0.000053780856,0.00025865895,0.00017838014,0.0036156499,0.02213717],"genre_scores_gemma":[0.91695845,0.000034460027,0.07950777,0.000041721218,0.000007676242,0.00007451377,0.00025602695,0.00022215833,0.0028972214],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9966184,0.0011266877,0.0001932026,0.000368596,0.0013068599,0.00038629706],"domain_scores_gemma":[0.9862132,0.0068061096,0.0009124151,0.0024178957,0.003064476,0.0005858636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046266494,0.00031323935,0.0003498184,0.0013898807,0.00052465586,0.001817713,0.0014253034,0.0009551907,0.0034391417],"category_scores_gemma":[0.01567254,0.00022494636,0.00047265278,0.0012326392,0.0009349978,0.0021694365,0.0015271469,0.0008058382,0.0004377196],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001771699,0.0022682874,0.1185583,0.00034440664,0.00014680298,0.0011667127,0.0022930626,0.15423542,0.04493135,0.05068579,0.0026131952,0.62098503],"study_design_scores_gemma":[0.00012706014,0.00083753746,0.026233653,0.000088639266,0.00013327258,0.000725743,0.00043090334,0.9168039,0.032513108,0.014409688,0.007638002,0.000058444453],"about_ca_topic_score_codex":0.008655168,"about_ca_topic_score_gemma":0.007214841,"teacher_disagreement_score":0.008655168,"about_ca_system_score_codex":0.0012240182,"about_ca_system_score_gemma":0.0024313396,"threshold_uncertainty_score":0.024468362},"labels":[],"label_agreement":null},{"id":"W2052416974","doi":"10.1109/qsic.2008.35","title":"Targeting Security Vulnerabilities: From Specification to Detection (Short Paper)","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Software security assurance; Security testing; Vulnerability management; Automaton; Vulnerability (computing); Software engineering; Chaining; Software; Computer security; Security information and event management; Information security; Vulnerability assessment; Programming language; Security service; Cloud computing security; Theoretical computer science; Operating system; Cloud computing","score_opus":0.026150733046258593,"score_gpt":0.24967659123988803,"score_spread":0.22352585819362944,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052416974","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0067350706,0.0023732807,0.98176426,0.0016549724,0.0003822942,0.000055211483,0.000059042428,0.0020696057,0.0049063372],"genre_scores_gemma":[0.31130585,0.008575727,0.6562619,0.0015045875,0.0010020315,0.00021860356,0.0006448276,0.0018198688,0.018666655],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99645823,0.0011588557,0.00032454098,0.0007295567,0.001077835,0.00025094277],"domain_scores_gemma":[0.99151164,0.0047954754,0.00046157764,0.001528111,0.0014240458,0.0002792035],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028205079,0.00089288945,0.0006155596,0.001070641,0.0004073581,0.0033111712,0.0012302733,0.0020000413,0.0058789393],"category_scores_gemma":[0.009991556,0.0010191969,0.0009842282,0.0011534222,0.0023384965,0.005347051,0.0016438742,0.0030272289,0.002491265],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023221718,0.00022778129,0.002746447,0.00092894846,0.000089434856,0.000794316,0.0011533266,0.019113187,0.028635355,0.21521717,0.020646814,0.7102151],"study_design_scores_gemma":[0.00011987357,0.0009200335,0.002539147,0.0010715143,0.00033397344,0.0034739736,0.00065374986,0.26365912,0.15994927,0.34136322,0.2256933,0.00022285566],"about_ca_topic_score_codex":0.0015313239,"about_ca_topic_score_gemma":0.0009725258,"teacher_disagreement_score":0.0058789393,"about_ca_system_score_codex":0.0009153725,"about_ca_system_score_gemma":0.0015121166,"threshold_uncertainty_score":0.01966697},"labels":[],"label_agreement":null},{"id":"W2053418084","doi":"10.1109/iceccs.2013.32","title":"Merging Test Models","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada; Concordia University","keywords":"Computer science; Unified Modeling Language; Test (biology); Model-based testing; Integration testing; Unit testing; Test Management Approach; Test case; Software engineering; Data mining; Programming language; Machine learning; Software; Software development","score_opus":0.02302779249184031,"score_gpt":0.2329466223616793,"score_spread":0.20991882986983898,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053418084","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018953918,0.0005613251,0.9588071,0.00054617756,0.00011634027,0.00049947295,0.0010972724,0.0074036894,0.01201479],"genre_scores_gemma":[0.3596848,0.0010075688,0.6153008,0.0007483468,0.0001247423,0.00074015575,0.011049934,0.003164854,0.008178882],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9849284,0.0047326065,0.0013983829,0.0015685905,0.006569201,0.00080272317],"domain_scores_gemma":[0.97553504,0.011275907,0.0012129936,0.007109361,0.0044645015,0.00040222227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00791669,0.0018468705,0.001389872,0.0037933798,0.00076856057,0.0045314077,0.0047003683,0.0023772968,0.008679284],"category_scores_gemma":[0.035589386,0.0012165334,0.0032442636,0.0030321043,0.0013121308,0.0068793423,0.0037979232,0.002699006,0.0022067179],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001088039,0.0005418585,0.012330874,0.0015285842,0.0007028212,0.0022723107,0.0020685676,0.23613024,0.024928141,0.18271789,0.015887529,0.51980317],"study_design_scores_gemma":[0.00018578353,0.00056102406,0.0023084406,0.00054335577,0.00077036716,0.0014155012,0.00052800786,0.6527901,0.052976638,0.19389442,0.09384673,0.00017956564],"about_ca_topic_score_codex":0.004088615,"about_ca_topic_score_gemma":0.0029647409,"teacher_disagreement_score":0.008679284,"about_ca_system_score_codex":0.0017733013,"about_ca_system_score_gemma":0.002875236,"threshold_uncertainty_score":0.04186797},"labels":[],"label_agreement":null},{"id":"W2053941028","doi":"10.5539/cis.v8n1p25","title":"Best Test Cases Selection Approach Using Genetic Algorithm","year":2015,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Sequence diagram; Computer science; Test case; Algorithm; Sequence (biology); Unified Modeling Language; Automatic test pattern generation; Graph; Relation (database); Path (computing); Genetic algorithm; Class diagram; Data mining; Theoretical computer science; Programming language; Software; Machine learning","score_opus":0.0559674720454713,"score_gpt":0.28562672178452114,"score_spread":0.22965924973904983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053941028","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03682467,0.0005432587,0.9546394,0.0003045781,0.000053615815,0.0003069505,0.00010834349,0.0013851827,0.005834032],"genre_scores_gemma":[0.35250255,0.0003827644,0.6425997,0.00027751818,0.000049018938,0.0005782514,0.0005794217,0.00021483426,0.0028157802],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978934,0.00080367364,0.00011271693,0.00033460493,0.00065124105,0.00020423252],"domain_scores_gemma":[0.9982205,0.0011244521,0.00011875556,0.00009704191,0.0003709843,0.00006815735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016741675,0.0015530038,0.0013587944,0.0053775776,0.0009152116,0.0013477036,0.00213508,0.001639286,0.003609129],"category_scores_gemma":[0.004568986,0.00074117735,0.0016710144,0.0018451238,0.00072448223,0.0012137503,0.0008760668,0.0009802412,0.00053337216],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016726427,0.00032486094,0.0045391805,0.00018775134,0.00026715046,0.00065130385,0.00020123404,0.6511748,0.0072020683,0.011698809,0.0027504358,0.32083526],"study_design_scores_gemma":[0.000054432257,0.00010581546,0.0005084761,0.000039474187,0.00009099142,0.00018683745,0.00006559019,0.98734015,0.0025012847,0.0072364965,0.0018499775,0.000020461759],"about_ca_topic_score_codex":0.0054118205,"about_ca_topic_score_gemma":0.0042611393,"teacher_disagreement_score":0.0054118205,"about_ca_system_score_codex":0.0011498097,"about_ca_system_score_gemma":0.0018282857,"threshold_uncertainty_score":0.012073696},"labels":[],"label_agreement":null},{"id":"W2054446920","doi":"10.1109/icstw.2014.39","title":"Murphy Tools: Utilizing Extracted GUI Models for Industrial Software Testing","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Graphical user interface; Software engineering; Graphical user interface testing; Keyword-driven testing; Process (computing); Model-based testing; Non-regression testing; Software; White-box testing; Manual testing; User interface; Test case; Software development; Software construction; Programming language; Machine learning; User interface design","score_opus":0.2133653832814831,"score_gpt":0.2982011252407428,"score_spread":0.08483574195925972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2054446920","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09848061,0.00023446327,0.8030606,0.00038016483,0.000051423318,0.0008202015,0.0015639041,0.08814593,0.007262759],"genre_scores_gemma":[0.3093879,0.0001856965,0.6792293,0.00011974829,0.000013530525,0.000616911,0.0038259076,0.0032530758,0.0033679148],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99716043,0.000822004,0.0002493318,0.00025766826,0.0013803904,0.00013012617],"domain_scores_gemma":[0.98607635,0.007837587,0.0015355241,0.0027097203,0.0016761752,0.00016464385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028995946,0.0013890894,0.0004755725,0.0025682112,0.00033919569,0.0015351117,0.002046423,0.00097547256,0.004281661],"category_scores_gemma":[0.023075072,0.0008345438,0.0007185326,0.0010034345,0.0005853756,0.0028465728,0.0017045666,0.0013012716,0.0012476417],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010287752,0.0012457451,0.023589192,0.0013141024,0.00020786906,0.0026040492,0.0029417842,0.13865374,0.10170763,0.020269064,0.022581019,0.68385696],"study_design_scores_gemma":[0.00030852703,0.0011890342,0.008084426,0.00040353608,0.00014706857,0.0014466641,0.00055624975,0.7835083,0.13706726,0.014796789,0.05229556,0.0001966993],"about_ca_topic_score_codex":0.0036100447,"about_ca_topic_score_gemma":0.0053320387,"teacher_disagreement_score":0.004281661,"about_ca_system_score_codex":0.00086539745,"about_ca_system_score_gemma":0.0015756493,"threshold_uncertainty_score":0.015334725},"labels":[],"label_agreement":null},{"id":"W2056339801","doi":"10.1002/j.2334-5837.2002.tb02569.x","title":"1.6.2 Formalizing a Structured Natural Language Requirements Specification Notation","year":2002,"lang":"en","type":"article","venue":"INCOSE International Symposium","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Notation; Computer science; Programming language; B-Method; Formal specification; Formal methods; Readability; Software engineering; Linguistics","score_opus":0.023913972960793797,"score_gpt":0.27798638223217603,"score_spread":0.25407240927138225,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056339801","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002465047,0.00005497394,0.9914043,0.0005170183,0.00006582155,0.00023842046,0.00031384014,0.0011741668,0.0037663186],"genre_scores_gemma":[0.034492493,0.00014152205,0.96105105,0.00022997295,0.00004045787,0.0004913052,0.0007487406,0.00018543052,0.0026190002],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99083906,0.004762713,0.0013583422,0.00082153105,0.0018540415,0.0003642865],"domain_scores_gemma":[0.9848087,0.008179189,0.0014981598,0.002210905,0.0030481566,0.00025504897],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012573198,0.0011456726,0.00049080333,0.00132846,0.00087191095,0.0039846613,0.002167163,0.0017818714,0.0050338157],"category_scores_gemma":[0.01628787,0.0009237992,0.0017589177,0.000918481,0.002638686,0.004682118,0.0018761284,0.0032166967,0.0019239218],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009969369,0.00011893282,0.000529525,0.0004739973,0.000028184058,0.0005861285,0.0019778355,0.024443215,0.011849888,0.913399,0.006285774,0.04020782],"study_design_scores_gemma":[0.00021476418,0.00030635137,0.00034144887,0.0007890481,0.00009512614,0.0015243402,0.0007357782,0.228815,0.048094142,0.37347057,0.3454652,0.00014823188],"about_ca_topic_score_codex":0.0020414733,"about_ca_topic_score_gemma":0.002581497,"teacher_disagreement_score":0.012573198,"about_ca_system_score_codex":0.0013123888,"about_ca_system_score_gemma":0.0040300866,"threshold_uncertainty_score":0.06649423},"labels":[],"label_agreement":null},{"id":"W2056488032","doi":"10.1145/1391984.1391985","title":"Unit-level test adequacy criteria for visual dataflow languages and a testing methodology","year":2008,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":92,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Dataflow; Context (archaeology); Unit testing; Visual programming language; Empirical research; Software engineering; Programming language; Software","score_opus":0.3232977272659157,"score_gpt":0.41639520446411027,"score_spread":0.09309747719819456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056488032","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0924143,0.00022208976,0.9037453,0.0001838755,0.000012396602,0.00025516242,0.00010529999,0.00076938054,0.002292308],"genre_scores_gemma":[0.6517582,0.000057913192,0.34695372,0.00008101921,0.000023303352,0.00041347998,0.00024574326,0.000115078954,0.00035150707],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9824097,0.008039705,0.0019729317,0.00097650156,0.0060489066,0.00055231777],"domain_scores_gemma":[0.89618325,0.08041725,0.0067043253,0.004858478,0.010725857,0.0011107944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010728257,0.0009186876,0.0007564116,0.0066180634,0.0006563935,0.0019363379,0.0015020302,0.0015145217,0.0017493347],"category_scores_gemma":[0.08693962,0.00039971937,0.0010738028,0.0018175408,0.0021377504,0.0024964712,0.0015668418,0.00093394046,0.00022096738],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011819414,0.0009115228,0.070972666,0.0011489677,0.00026975645,0.001300904,0.0018657002,0.22412844,0.07255206,0.14304198,0.0028473097,0.47977868],"study_design_scores_gemma":[0.00011847433,0.00088182854,0.0071166647,0.00019748445,0.000080428545,0.0010196358,0.00040821446,0.90512645,0.038119003,0.04430988,0.002561794,0.000060204307],"about_ca_topic_score_codex":0.0018150571,"about_ca_topic_score_gemma":0.0018517158,"teacher_disagreement_score":0.010728257,"about_ca_system_score_codex":0.0012778394,"about_ca_system_score_gemma":0.0014907888,"threshold_uncertainty_score":0.056737125},"labels":[],"label_agreement":null},{"id":"W2056640446","doi":"10.1016/j.infsof.2009.06.006","title":"Using machine learning to refine Category-Partition test specifications and test suites","year":2009,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Dependability; Software engineering; Test Management Approach; Partition (number theory); Test (biology); Context (archaeology); Test strategy; Code coverage; Machine learning; Software testing; Test case; Process (computing); Artificial intelligence; Software; Programming language; Software development; Software construction","score_opus":0.03283553674994307,"score_gpt":0.26684519881008295,"score_spread":0.23400966206013987,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056640446","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059513774,0.0002845202,0.9327241,0.00012547833,0.00003914297,0.00015618891,0.00032677228,0.005726001,0.0011039956],"genre_scores_gemma":[0.39530534,0.00011992924,0.6000442,0.00012732758,0.000024453442,0.00017396946,0.0023039111,0.00091819966,0.0009826757],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9944384,0.0017428334,0.00044871386,0.0008688683,0.0020455683,0.000455694],"domain_scores_gemma":[0.95608634,0.028841058,0.0024850895,0.0049644117,0.007174145,0.00044898613],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004020345,0.0016291707,0.0016878303,0.0055146012,0.0008174949,0.0016907814,0.0026066867,0.0012005828,0.002918421],"category_scores_gemma":[0.039654363,0.0011080982,0.0023281712,0.0020967447,0.0014583986,0.0026255834,0.0020007142,0.0022987516,0.0006249362],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060375454,0.00035715353,0.012252369,0.00053355214,0.00021923453,0.0004912703,0.0005732608,0.44053012,0.01612769,0.01686874,0.003398292,0.50804454],"study_design_scores_gemma":[0.000049041417,0.000114258095,0.0013043154,0.00007116849,0.00007324348,0.00011253839,0.00008178962,0.961361,0.007932472,0.027543496,0.0013324752,0.000024127832],"about_ca_topic_score_codex":0.013360415,"about_ca_topic_score_gemma":0.020484487,"teacher_disagreement_score":0.013360415,"about_ca_system_score_codex":0.0019423586,"about_ca_system_score_gemma":0.0036135472,"threshold_uncertainty_score":0.026565313},"labels":[],"label_agreement":null},{"id":"W2056904432","doi":"10.1109/issre.2014.14","title":"Multi-objective Construction of an Entire Adequate Test Suite for an EFSM","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Test suite; Computer science; Test (biology); Suite; Extended finite-state machine; Test case; Finite-state machine; Code coverage; Model-based testing; Reliability engineering; Algorithm; Machine learning; Programming language; Engineering; Software","score_opus":0.029913801385492645,"score_gpt":0.2933847128231663,"score_spread":0.26347091143767365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056904432","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050859302,0.000083182844,0.94690824,0.000106040985,0.000009936841,0.00013482453,0.000071078844,0.0004939707,0.0013333196],"genre_scores_gemma":[0.32210368,0.00007585379,0.6763488,0.00006485155,0.00001029547,0.00035650757,0.0003003302,0.00013307184,0.00060664286],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99853826,0.00065257517,0.0000777379,0.00015970111,0.00044928855,0.00012252582],"domain_scores_gemma":[0.99617106,0.0027368146,0.00030136885,0.00022337536,0.00047244702,0.000094909],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021858998,0.0010036229,0.0007024736,0.0017478573,0.0003133848,0.0006293941,0.00087220717,0.0007922242,0.0014643241],"category_scores_gemma":[0.0065548774,0.00039240505,0.0010804964,0.00052511116,0.00077828654,0.00074738386,0.00078902824,0.00088300765,0.00019739759],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009453389,0.00017079066,0.003377342,0.00025875287,0.000096754105,0.00029928508,0.00015181082,0.8587765,0.014078842,0.0111951195,0.00054794026,0.11095233],"study_design_scores_gemma":[0.00002130603,0.00019729837,0.0007433369,0.00004014021,0.000037355538,0.00012782325,0.00004316729,0.9852224,0.005734597,0.006927301,0.0008931228,0.000012244688],"about_ca_topic_score_codex":0.001068089,"about_ca_topic_score_gemma":0.0012013448,"teacher_disagreement_score":0.0021858998,"about_ca_system_score_codex":0.0006166788,"about_ca_system_score_gemma":0.0015085564,"threshold_uncertainty_score":0.011560261},"labels":[],"label_agreement":null},{"id":"W2057272331","doi":"10.1109/issrew.2014.9","title":"Trace Reduction and Pattern Analysis to Assist Debugging in Model-Based Testing","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"TRACE (psycholinguistics); Debugging; Computer science; Root cause; Reduction (mathematics); Test (biology); Model-based testing; Test case; Root cause analysis; Software bug; Data mining; Embedded system; Programming language; Reliability engineering; Machine learning; Software; Engineering","score_opus":0.029121904949418385,"score_gpt":0.26932076576125136,"score_spread":0.24019886081183298,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2057272331","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015843246,0.00008095809,0.98102635,0.00014754795,0.000014389412,0.000057698286,0.000053534273,0.00204247,0.0007337631],"genre_scores_gemma":[0.21843271,0.00018233749,0.77932715,0.0000920453,0.000015478952,0.00011702521,0.0003061568,0.00035100715,0.0011762219],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99834776,0.00061991974,0.00010892103,0.00016099786,0.00068170595,0.00008073693],"domain_scores_gemma":[0.99538016,0.002768554,0.00029490024,0.0009313604,0.0005646293,0.000060539005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012476381,0.00094812026,0.00052749476,0.0018069545,0.00036281362,0.00066037825,0.0013204658,0.0006616517,0.0019344691],"category_scores_gemma":[0.009547042,0.00038863719,0.00078587484,0.0015634961,0.000637265,0.0016582712,0.00078605005,0.001099255,0.0003402118],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024286634,0.00046346834,0.0068940674,0.00031065758,0.000083036204,0.00075614225,0.0005399282,0.15894084,0.059452444,0.03881986,0.004127834,0.72936875],"study_design_scores_gemma":[0.0000442697,0.00025050397,0.0013894521,0.00005209275,0.000051313993,0.00078244205,0.000092681345,0.91032743,0.047054525,0.03350541,0.0064151124,0.000034735156],"about_ca_topic_score_codex":0.0018312355,"about_ca_topic_score_gemma":0.0026530456,"teacher_disagreement_score":0.0019344691,"about_ca_system_score_codex":0.00041385763,"about_ca_system_score_gemma":0.0008683363,"threshold_uncertainty_score":0.006598234},"labels":[],"label_agreement":null},{"id":"W2058792175","doi":"10.1016/j.artint.2005.09.002","title":"Finding optimal satisficing strategies for and-or trees","year":2005,"lang":"en","type":"article","venue":"Artificial Intelligence","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of Alberta","funders":"","keywords":"Probabilistic logic; Tree (set theory); Set (abstract data type); Node (physics); Computer science; Outcome (game theory); Satisficing; Class (philosophy); Mathematical optimization; Task (project management); Mathematics; Test (biology); Algorithm; Artificial intelligence; Combinatorics; Mathematical economics","score_opus":0.1111722750895888,"score_gpt":0.3551924202138077,"score_spread":0.2440201451242189,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2058792175","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24767894,0.0013254884,0.7307249,0.001500203,0.000104911334,0.00054983184,0.000820693,0.002344673,0.014950485],"genre_scores_gemma":[0.55956066,0.0005534412,0.4339123,0.000459943,0.00004419271,0.00019173752,0.0010327402,0.0005722511,0.0036727302],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99796057,0.0007931411,0.00017038695,0.0003597488,0.00031276883,0.0004034242],"domain_scores_gemma":[0.98693234,0.011150697,0.00043081286,0.0007053852,0.0005311449,0.0002495409],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027010106,0.0015645771,0.0017334413,0.0014608566,0.0009453686,0.0022676478,0.0022756886,0.0020890594,0.006416251],"category_scores_gemma":[0.015813092,0.00144753,0.0016053336,0.0013287953,0.001554689,0.0058980375,0.0014674802,0.0026103237,0.0007713961],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015491669,0.0007812786,0.004967705,0.0016681603,0.0003306123,0.00057979376,0.0015911307,0.27669242,0.01560396,0.20768705,0.017290283,0.47125852],"study_design_scores_gemma":[0.00025179514,0.00025245795,0.00049839774,0.00015036644,0.00023142662,0.00019516885,0.0006172233,0.6072464,0.0071491287,0.38046905,0.0028875696,0.000051061288],"about_ca_topic_score_codex":0.0025316675,"about_ca_topic_score_gemma":0.0059445878,"teacher_disagreement_score":0.006416251,"about_ca_system_score_codex":0.0016906108,"about_ca_system_score_gemma":0.0021397623,"threshold_uncertainty_score":0.021464527},"labels":[],"label_agreement":null},{"id":"W2059174184","doi":"10.5555/2666527.2666536","title":"Predicting mutation score using source code and test suite metrics","year":2012,"lang":"en","type":"article","venue":"e-scholar@UOIT (University of Ontario Institute of Technology)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Test suite; Mutation; Computer science; Mutation testing; Suite; Source code; Code (set theory); Process (computing); Test case; Machine learning; Programming language; Set (abstract data type); Biology; Genetics","score_opus":0.030706705246584875,"score_gpt":0.22843711999654606,"score_spread":0.1977304147499612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2059174184","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92220455,0.00037316378,0.07130553,0.00021556672,0.000028228887,0.00011828381,0.0014169597,0.0027931104,0.0015445873],"genre_scores_gemma":[0.95184445,0.00010134764,0.044588886,0.000025996822,0.000017772627,0.000055690864,0.0027119024,0.000069570444,0.0005844125],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977633,0.0005190994,0.00022228535,0.00041570805,0.00092979084,0.00014988029],"domain_scores_gemma":[0.9812207,0.01025911,0.002782163,0.0008955727,0.0042350465,0.0006074652],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026495429,0.001199849,0.0006369401,0.009225321,0.0002565261,0.0011035296,0.0007244335,0.0010576441,0.0005372395],"category_scores_gemma":[0.020881087,0.00025366037,0.00058809284,0.0029349867,0.00029560912,0.001196232,0.00051195157,0.0006046837,0.000399921],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025577727,0.0007072704,0.47529906,0.00017372091,0.0002820196,0.00028700219,0.000122051104,0.17864606,0.013875625,0.0008659016,0.0030168681,0.32646868],"study_design_scores_gemma":[0.000012731435,0.00013273573,0.041068796,0.000010500816,0.000033958528,0.0000828081,0.000030588348,0.9529048,0.004731893,0.00057720277,0.00039302727,0.00002086895],"about_ca_topic_score_codex":0.0078951735,"about_ca_topic_score_gemma":0.012790759,"teacher_disagreement_score":0.009225321,"about_ca_system_score_codex":0.0010199221,"about_ca_system_score_gemma":0.00081625534,"threshold_uncertainty_score":0.015698433},"labels":[],"label_agreement":null},{"id":"W2059896025","doi":"10.1145/1218563.1218575","title":"Debugging with control-flow breakpoints","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Debugging; Breakpoint; Control flow; Debugger; Programming language; Guard (computer science)","score_opus":0.0070093457028262305,"score_gpt":0.22456384008184785,"score_spread":0.2175544943790216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2059896025","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037514456,0.00034741272,0.9417922,0.00025388427,0.000059151724,0.00016207689,0.0002352939,0.016857382,0.0027781674],"genre_scores_gemma":[0.47709432,0.00037264204,0.5165496,0.00023554923,0.000031377313,0.00022621053,0.0006326628,0.0027010683,0.0021565284],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99415576,0.00210069,0.0006024731,0.0010764501,0.0016713568,0.00039321714],"domain_scores_gemma":[0.9727625,0.017654276,0.0019805394,0.0053856364,0.0019372377,0.00027982297],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068567893,0.0015448189,0.00072205294,0.002395093,0.0005726563,0.0019995777,0.0019643884,0.0016517484,0.0034705934],"category_scores_gemma":[0.0399916,0.0008440281,0.00078097597,0.0013180338,0.0017333538,0.004785744,0.0026467873,0.0022580752,0.0006180156],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011341819,0.0004205642,0.027407277,0.0011348808,0.00015190431,0.0018546865,0.0044150264,0.09634902,0.036105663,0.082274616,0.010890893,0.73786134],"study_design_scores_gemma":[0.0004501346,0.00087248726,0.009439761,0.0011648795,0.00032364397,0.0034631302,0.0009953754,0.48343068,0.17282674,0.20205078,0.12455483,0.00042761603],"about_ca_topic_score_codex":0.0021550797,"about_ca_topic_score_gemma":0.0016985161,"teacher_disagreement_score":0.0068567893,"about_ca_system_score_codex":0.00079485116,"about_ca_system_score_gemma":0.0011291401,"threshold_uncertainty_score":0.03626257},"labels":[],"label_agreement":null},{"id":"W2060107892","doi":"10.1109/scam.2010.26","title":"How Good is Static Analysis at Finding Concurrency Bugs?","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Concurrency; Interleaving; Static analysis; Thread (computing); Java; Software bug; Debugging; Spurious relationship; Programming language; Software; Operating system; Parallel computing","score_opus":0.023249236838585617,"score_gpt":0.2784057360787849,"score_spread":0.2551564992401993,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2060107892","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.73103064,0.011276689,0.2241984,0.008347528,0.00046425953,0.000216497,0.00033840278,0.005902049,0.018225612],"genre_scores_gemma":[0.9310397,0.001984859,0.06501948,0.00048145626,0.0001359067,0.000037744354,0.0001285489,0.0003603586,0.0008118433],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.97322154,0.0115046,0.0014655743,0.0026421566,0.009972978,0.0011932489],"domain_scores_gemma":[0.77811694,0.17441964,0.010364147,0.015242112,0.019821381,0.0020358313],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018960958,0.0013969557,0.0022662643,0.0067720986,0.0017005444,0.004491993,0.0021630486,0.0024912509,0.0018345725],"category_scores_gemma":[0.14990969,0.0011254975,0.001430951,0.0040936065,0.0030505701,0.011465792,0.0016746043,0.0010985364,0.001271536],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011656241,0.0006442418,0.25092617,0.0011334552,0.0007316762,0.00035970804,0.0021878101,0.011684061,0.02252979,0.0031336602,0.0027479455,0.7027557],"study_design_scores_gemma":[0.0005620642,0.011509455,0.4624751,0.0025749796,0.0039826836,0.007536162,0.014406968,0.29071474,0.10180629,0.07963059,0.023401458,0.0013995374],"about_ca_topic_score_codex":0.0049002855,"about_ca_topic_score_gemma":0.006477379,"teacher_disagreement_score":0.018960958,"about_ca_system_score_codex":0.0011689556,"about_ca_system_score_gemma":0.0019875614,"threshold_uncertainty_score":0.10027629},"labels":[],"label_agreement":null},{"id":"W2061588280","doi":"10.1109/icst.2014.27","title":"Automated Bug Finding in Video Games: A Case Study for Runtime Monitoring","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Computer science; Video game; Video game development; Game programming; Runtime verification; Process (computing); Game development tool; Instrumentation (computer programming); Game testing; Game Developer; Software; State (computer science); Sample (material); Programming language; Real-time computing; Game design; Formal verification; Human–computer interaction; Game design document; Game art design; Multimedia","score_opus":0.04248516793564986,"score_gpt":0.33574468143082703,"score_spread":0.29325951349517715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061588280","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.964895,0.00026398426,0.030917985,0.00021634281,0.000021369176,0.000270942,0.00040524665,0.0012320537,0.0017769979],"genre_scores_gemma":[0.9561634,0.00010052598,0.04226915,0.000057230747,0.000010637002,0.000074413525,0.00036244633,0.00014042831,0.0008218556],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99695575,0.0011931877,0.00020377924,0.0005778854,0.0008087745,0.00026058443],"domain_scores_gemma":[0.9821425,0.011867659,0.0017010754,0.0019317666,0.0015574777,0.00079950056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021203263,0.0009360683,0.00048490963,0.0015913532,0.0010085208,0.00094895443,0.001776827,0.0012691311,0.0006509939],"category_scores_gemma":[0.017621461,0.00041425688,0.000535862,0.001174401,0.0012740976,0.0010308621,0.00091700326,0.0010845156,0.00022559546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032957788,0.0057679145,0.33866653,0.0017125194,0.00054903224,0.03127762,0.024359815,0.0765334,0.122095875,0.012861636,0.011355353,0.3715245],"study_design_scores_gemma":[0.00077454135,0.0050633736,0.29736084,0.00036400408,0.000539631,0.018844767,0.008756914,0.45582908,0.16631575,0.010608429,0.0352166,0.0003261689],"about_ca_topic_score_codex":0.010002529,"about_ca_topic_score_gemma":0.014971792,"teacher_disagreement_score":0.010002529,"about_ca_system_score_codex":0.0008815226,"about_ca_system_score_gemma":0.0007496233,"threshold_uncertainty_score":0.01988858},"labels":[],"label_agreement":null},{"id":"W2063462461","doi":"10.4018/jdm.2002040103","title":"Regression Testing of Database Applications","year":2002,"lang":"en","type":"article","venue":"Journal of Database Management","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Regression testing; Data mining; Control flow graph; Test case; Regression analysis; Database; Machine learning; Theoretical computer science; Programming language; Software; Software system","score_opus":0.06318055350265096,"score_gpt":0.2895761658878142,"score_spread":0.22639561238516326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2063462461","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4706854,0.000562785,0.5142729,0.00043683138,0.00006176791,0.0003445778,0.00030872884,0.007852531,0.0054744426],"genre_scores_gemma":[0.90606207,0.0001904132,0.09162062,0.00011978634,0.000023231765,0.0001263179,0.0004591361,0.0003243993,0.0010740083],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9940546,0.002795377,0.00029343995,0.000584915,0.0018734086,0.00039829235],"domain_scores_gemma":[0.97460496,0.018450785,0.001431498,0.0024362449,0.0028291352,0.00024739932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029044556,0.00078688725,0.00047781537,0.0015731832,0.0003121679,0.0006554255,0.0016076764,0.0005054033,0.0015025323],"category_scores_gemma":[0.023494035,0.000282076,0.0006350148,0.00091471174,0.0004658281,0.0011473387,0.0008022044,0.000826138,0.0003972413],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009904917,0.0011488354,0.039300475,0.0006121495,0.00022330572,0.001176969,0.00055855187,0.14288391,0.14256732,0.012908067,0.0033489661,0.654281],"study_design_scores_gemma":[0.00011470843,0.0014630193,0.017600762,0.00008741358,0.00013710994,0.0010246048,0.000200576,0.7943872,0.16807623,0.011498554,0.0053566797,0.000053109958],"about_ca_topic_score_codex":0.0016859103,"about_ca_topic_score_gemma":0.0012255295,"teacher_disagreement_score":0.0029044556,"about_ca_system_score_codex":0.0005092866,"about_ca_system_score_gemma":0.0004574635,"threshold_uncertainty_score":0.015360415},"labels":[],"label_agreement":null},{"id":"W2065604113","doi":"10.1109/mra.2008.931632","title":"The robotics experience","year":2009,"lang":"en","type":"article","venue":"IEEE Robotics & Automation Magazine","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada","funders":"","keywords":"Reuse; Robotics; Reusability; Software engineering; Variety (cybernetics); Artificial intelligence; Software; Computer science; Robot; Systems engineering; Engineering; Programming language","score_opus":0.021030628607017297,"score_gpt":0.282433840506703,"score_spread":0.2614032118996857,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2065604113","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016916282,0.10052978,0.014011225,0.12414396,0.006647776,0.00007100847,0.00030846085,0.000572956,0.7367986],"genre_scores_gemma":[0.26836038,0.15065953,0.017409965,0.045325473,0.004288055,0.0002163858,0.0005451827,0.00057778286,0.51261723],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9966198,0.0011778565,0.00014058503,0.0005363861,0.0010124531,0.0005128676],"domain_scores_gemma":[0.99692804,0.00045878254,0.00017230232,0.00023685176,0.00041862216,0.0017854723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039518,0.0009834897,0.00073693827,0.001442892,0.0048367856,0.006810837,0.001892386,0.0038757792,0.06461883],"category_scores_gemma":[0.0059063695,0.0003841906,0.00045500696,0.0014717977,0.0069626183,0.009345259,0.00707268,0.0039782566,0.028863978],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000100969475,0.00022225553,0.0020874655,0.00066069886,0.000025529544,0.0012589755,0.017421043,0.00057562615,0.0005311797,0.19365528,0.46550462,0.3179564],"study_design_scores_gemma":[0.0000037556965,0.00003865343,0.00037757505,0.00023041334,0.0000019841723,0.0010943145,0.0024611512,0.000042098713,0.000046190904,0.009774922,0.98591655,0.000012224789],"about_ca_topic_score_codex":0.0034308953,"about_ca_topic_score_gemma":0.010335485,"teacher_disagreement_score":0.06461883,"about_ca_system_score_codex":0.0034820887,"about_ca_system_score_gemma":0.0050699464,"threshold_uncertainty_score":0.2161715},"labels":[],"label_agreement":null},{"id":"W2067531935","doi":"10.1093/comjnl/bxu113","title":"Generalizing the DS-Methods for Testing Non-Deterministic FSMs","year":2014,"lang":"en","type":"article","venue":"The Computer Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Algorithm","score_opus":0.05822499298085235,"score_gpt":0.35446574423089033,"score_spread":0.296240751250038,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2067531935","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0043367674,0.00012302035,0.994166,0.000046984005,0.000023295905,0.00006065072,0.000033459168,0.00046257814,0.00074725744],"genre_scores_gemma":[0.28045,0.0004724045,0.7135263,0.00028490723,0.00011536861,0.00054309965,0.00042095577,0.00039343286,0.0037935372],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9967001,0.0009692345,0.00031431884,0.0007890841,0.0010684431,0.00015890563],"domain_scores_gemma":[0.9884899,0.0070595043,0.0007223009,0.0022358156,0.0012848658,0.00020754864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020925824,0.0012564306,0.0007481859,0.0016037776,0.00031166148,0.0008592465,0.001468563,0.00080019986,0.0020848985],"category_scores_gemma":[0.011248877,0.00041064472,0.0014328975,0.0007771701,0.0024181884,0.0020663626,0.0019462856,0.0018581094,0.0006062606],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045632652,0.00013954574,0.0052043335,0.001126171,0.00019346975,0.00030139176,0.0008018307,0.15311454,0.04985647,0.35572714,0.0014403795,0.43163842],"study_design_scores_gemma":[0.00010438889,0.00047309717,0.0011691387,0.00018039554,0.00010966143,0.00058257225,0.00011585649,0.60513353,0.06479246,0.29836482,0.028892258,0.00008179892],"about_ca_topic_score_codex":0.0015577942,"about_ca_topic_score_gemma":0.0012395013,"teacher_disagreement_score":0.0020925824,"about_ca_system_score_codex":0.00095102127,"about_ca_system_score_gemma":0.0014327461,"threshold_uncertainty_score":0.011066794},"labels":[],"label_agreement":null},{"id":"W2067617772","doi":"10.1109/tse.2014.2363479","title":"Instance Generator and Problem Representation to Improve Object Oriented Code Coverage","year":2014,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":86,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Unit testing; Test case; Programming language; Code generation; Java; Code coverage; Object-oriented programming; Data structure; Source code; Generator (circuit theory); Software; Theoretical computer science; Machine learning; Operating system","score_opus":0.009565537506992744,"score_gpt":0.23054314654182123,"score_spread":0.2209776090348285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2067617772","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061452817,0.0002558603,0.9303147,0.00041040045,0.000043555832,0.0003015467,0.00027200987,0.0042056898,0.0027433787],"genre_scores_gemma":[0.30875137,0.00014666683,0.6867939,0.00019079843,0.00004021052,0.000581838,0.0011326323,0.00071254605,0.0016499636],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980094,0.00091809797,0.00013072437,0.00026941416,0.00052619755,0.00014613442],"domain_scores_gemma":[0.99071336,0.007262501,0.00036670253,0.00081413565,0.00070327363,0.00013998128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023281644,0.0010175756,0.00083197403,0.0015311969,0.00033179036,0.0010157257,0.0016761124,0.0012188375,0.0034384422],"category_scores_gemma":[0.016021011,0.00047048883,0.0011986118,0.0011815666,0.00088339136,0.0014141997,0.0016391369,0.0012809803,0.0005158788],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003591938,0.0005333858,0.006147843,0.0005472008,0.000110511515,0.00045382237,0.00041982537,0.63214445,0.012555314,0.037743483,0.008172878,0.30081216],"study_design_scores_gemma":[0.000078819634,0.000060693546,0.00024681306,0.000020275153,0.000025980731,0.000088635454,0.000032182463,0.98330265,0.004014054,0.009934347,0.0021875813,0.000007931818],"about_ca_topic_score_codex":0.0014607677,"about_ca_topic_score_gemma":0.0017441,"teacher_disagreement_score":0.0034384422,"about_ca_system_score_codex":0.00075841846,"about_ca_system_score_gemma":0.0013874777,"threshold_uncertainty_score":0.012312591},"labels":[],"label_agreement":null},{"id":"W2068493323","doi":"10.1109/tse.2011.56","title":"Size-Constrained Regression Test Case Selection Using Multicriteria Optimization","year":2011,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Test suite; Test case; Consistency (knowledge bases); Integer programming; Mathematical optimization; Task (project management); Constraint (computer-aided design); Selection (genetic algorithm); Regression testing; Greedy algorithm; Linear programming; Relaxation (psychology); Software; Algorithm; Machine learning; Regression analysis; Artificial intelligence; Software system; Mathematics","score_opus":0.03160363037280706,"score_gpt":0.24695615889016168,"score_spread":0.21535252851735462,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2068493323","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042217884,0.0002500273,0.9535501,0.00018956552,0.00001864077,0.00027929642,0.00005645239,0.0011299348,0.0023079638],"genre_scores_gemma":[0.50886685,0.00013158054,0.48781067,0.00019684482,0.000035145255,0.000607582,0.0003522329,0.00030591027,0.0016932025],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99636984,0.001880121,0.000155738,0.00046025356,0.00080767577,0.000326385],"domain_scores_gemma":[0.98945516,0.008016248,0.0009346336,0.00050066254,0.00092964614,0.00016358956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003340231,0.0017932216,0.0021227065,0.0025780003,0.00045570367,0.0009586763,0.0019414583,0.0010370733,0.0022639323],"category_scores_gemma":[0.012145688,0.00081527745,0.0013888077,0.0014915229,0.0008785232,0.0010260463,0.00104371,0.0011201613,0.00039728318],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018284851,0.00015764721,0.0014335392,0.00008623061,0.000092613016,0.00010823306,0.000049978273,0.88577604,0.0054393634,0.0028116147,0.0008575861,0.10300437],"study_design_scores_gemma":[0.000030175184,0.00006433009,0.00021401522,0.0000059557965,0.00001546945,0.000025702788,0.000008331274,0.99693537,0.0013848626,0.0011398932,0.00016995223,0.000005929534],"about_ca_topic_score_codex":0.0040839,"about_ca_topic_score_gemma":0.004639899,"teacher_disagreement_score":0.0040839,"about_ca_system_score_codex":0.0012796273,"about_ca_system_score_gemma":0.0018423954,"threshold_uncertainty_score":0.017665029},"labels":[],"label_agreement":null},{"id":"W2069169007","doi":"10.1109/wse.2012.6320526","title":"Visual testing of Graphical User Interfaces: An exploratory study towards systematic definitions and approaches","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Graphical user interface; Computer science; Graphical user interface testing; Human–computer interaction; User interface; Visualization; User interface design; User experience design; Artificial intelligence; Programming language","score_opus":0.22841786623247345,"score_gpt":0.31590808405690346,"score_spread":0.08749021782443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2069169007","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8030003,0.011457379,0.16668715,0.002761069,0.000047881604,0.0025605287,0.00019258673,0.00020561997,0.013087488],"genre_scores_gemma":[0.8988955,0.0042858855,0.09379148,0.00060292974,0.000019585214,0.0012759982,0.00016428067,0.00008256236,0.0008817853],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9251545,0.05964667,0.0038325542,0.002374541,0.007812857,0.0011788935],"domain_scores_gemma":[0.6916152,0.26064688,0.017366106,0.01222075,0.016569827,0.0015811274],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.038675215,0.0008796976,0.0008285689,0.007470798,0.002113699,0.0044851494,0.0027601374,0.0016426626,0.0008141059],"category_scores_gemma":[0.13735645,0.0007978407,0.00061102863,0.0048984583,0.008489107,0.0073582083,0.005952691,0.0024407662,0.00015118589],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025223164,0.0014375962,0.072003126,0.005005461,0.000076633194,0.0014511615,0.46779895,0.0016053865,0.012722938,0.035833444,0.0011179212,0.40069526],"study_design_scores_gemma":[0.00026949993,0.0043512136,0.123884864,0.022322243,0.00027605804,0.0068216156,0.64707196,0.013496172,0.02816086,0.054611,0.09841506,0.0003193945],"about_ca_topic_score_codex":0.002986981,"about_ca_topic_score_gemma":0.004284854,"teacher_disagreement_score":0.038675215,"about_ca_system_score_codex":0.0037352198,"about_ca_system_score_gemma":0.0070116,"threshold_uncertainty_score":0.20453656},"labels":[],"label_agreement":null},{"id":"W2070233995","doi":"10.1145/2678022","title":"Dynamically Instrumenting the QEMU Emulator for Linux Process Trace Generation with the GDB Debugger","year":2014,"lang":"en","type":"article","venue":"ACM Transactions on Embedded Computing Systems","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Debugging; TRACE (psycholinguistics); Emulation; Tracing; Operating system; Process (computing); Debugger; Software; Embedded system","score_opus":0.020906165101975738,"score_gpt":0.2672740377382368,"score_spread":0.2463678726362611,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070233995","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2601056,0.00023262121,0.61776793,0.0003381194,0.00011739843,0.0002368871,0.00043643525,0.11685563,0.0039093476],"genre_scores_gemma":[0.79928315,0.00007904232,0.1935733,0.00019877148,0.00002232193,0.0002126151,0.0005350355,0.0032422652,0.0028535498],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986203,0.00030617104,0.00008349793,0.00028329832,0.0005362989,0.00017056122],"domain_scores_gemma":[0.996507,0.0010656028,0.00032421193,0.0014971505,0.0004497268,0.00015633208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013095133,0.00074599806,0.00041291892,0.0013545686,0.00030745598,0.0009640597,0.0016131073,0.0005296627,0.0032860613],"category_scores_gemma":[0.008425748,0.00052307197,0.00020492404,0.0006388997,0.00050066004,0.0016640075,0.0015284256,0.0011946071,0.0011564399],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020968083,0.0005759611,0.034818213,0.00031870036,0.000104458166,0.0005017484,0.0015633298,0.01863586,0.27160662,0.008843081,0.016472572,0.64446276],"study_design_scores_gemma":[0.00020356843,0.00071226864,0.014808528,0.00008740995,0.00006478118,0.00062741863,0.00021022398,0.3506563,0.5889014,0.0039814766,0.039590526,0.00015610203],"about_ca_topic_score_codex":0.0012986147,"about_ca_topic_score_gemma":0.0013854925,"teacher_disagreement_score":0.0032860613,"about_ca_system_score_codex":0.0006212634,"about_ca_system_score_gemma":0.000949351,"threshold_uncertainty_score":0.010993004},"labels":[],"label_agreement":null},{"id":"W2070633450","doi":"10.1109/icstw.2013.31","title":"A Method and Tool for Test Optimization for Automotive Controllers","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Automotive industry; Computer science; Test (biology); Automotive engineering; Control engineering; Engineering; Aerospace engineering","score_opus":0.014788650045673982,"score_gpt":0.28235754149414083,"score_spread":0.26756889144846685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070633450","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00042771056,0.00005048458,0.9913139,0.00005108148,0.000018615843,0.00006966911,0.00007351946,0.0073525,0.00064250577],"genre_scores_gemma":[0.019612968,0.00012248008,0.9753964,0.000081568236,0.00003074623,0.00039130935,0.00041225733,0.0020455674,0.0019068368],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99723023,0.000584792,0.00026444776,0.00047412264,0.0012756805,0.00017067241],"domain_scores_gemma":[0.99671715,0.002044781,0.00024798783,0.00054882106,0.00036975922,0.00007146622],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021069602,0.0021422266,0.0010561172,0.002561095,0.0007928458,0.0019090407,0.0027146242,0.0015766773,0.013141416],"category_scores_gemma":[0.008781594,0.0014618455,0.0022971905,0.0014678396,0.0017057811,0.0023154588,0.0020575419,0.0029277217,0.0034898312],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021683107,0.00024833428,0.0013660016,0.0010234871,0.00018855667,0.00096382364,0.0004826707,0.06554379,0.036909383,0.12025678,0.024550306,0.74824995],"study_design_scores_gemma":[0.00028484274,0.00030825083,0.00094456726,0.0004179353,0.00015017875,0.002622537,0.00011783631,0.63393396,0.06210784,0.122412615,0.17651497,0.00018441447],"about_ca_topic_score_codex":0.0017492308,"about_ca_topic_score_gemma":0.0016872582,"teacher_disagreement_score":0.013141416,"about_ca_system_score_codex":0.000832743,"about_ca_system_score_gemma":0.0018341778,"threshold_uncertainty_score":0.04396242},"labels":[],"label_agreement":null},{"id":"W2070655751","doi":"10.5555/2819261.2819271","title":"Adaptive random testing by static partitioning","year":2015,"lang":"en","type":"article","venue":"Automation of Software Test","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Random testing; Computer science; Overhead (engineering); Point (geometry); Algorithm; Test case; Parallel computing; Distributed computing; Theoretical computer science; Mathematics; Machine learning","score_opus":0.048154716995931886,"score_gpt":0.2663096841822027,"score_spread":0.2181549671862708,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070655751","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0193952,0.00017357903,0.9764009,0.00009832773,0.0000338147,0.00008587354,0.000028455557,0.0011577493,0.0026261895],"genre_scores_gemma":[0.5412027,0.000191325,0.45482045,0.00017341142,0.000047288384,0.00025217788,0.0001850986,0.00043230978,0.0026951486],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977362,0.0007534665,0.00011692604,0.00038232945,0.0007868087,0.00022428946],"domain_scores_gemma":[0.99504447,0.0023570335,0.0004141031,0.0010796341,0.0009631102,0.00014166567],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013495989,0.0009823682,0.0007886994,0.0014029959,0.00060699054,0.0008880469,0.0018912606,0.00068087154,0.002397967],"category_scores_gemma":[0.007875104,0.0004589129,0.0006944378,0.0008766513,0.0010240508,0.0018591391,0.0015069944,0.00079273246,0.00079095055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005163273,0.00019278667,0.0044901706,0.00026086997,0.00009542358,0.000440102,0.00030915,0.36796066,0.04428661,0.04686249,0.003991455,0.53059393],"study_design_scores_gemma":[0.000060497816,0.00020524295,0.00087651995,0.000033172266,0.000044544904,0.00059082045,0.000055869856,0.94583225,0.019907774,0.027937995,0.004413676,0.00004159437],"about_ca_topic_score_codex":0.0016428658,"about_ca_topic_score_gemma":0.001677903,"teacher_disagreement_score":0.002397967,"about_ca_system_score_codex":0.00073226506,"about_ca_system_score_gemma":0.0009944007,"threshold_uncertainty_score":0.008021951},"labels":[],"label_agreement":null},{"id":"W2070960202","doi":"10.1109/cec.2010.5586002","title":"A binary Particle Swarm Optimization approach to fault diagnosis in parallel and distributed systems","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Particle swarm optimization; Heuristic; Set (abstract data type); Task (project management); Binary number; Fault (geology); Identification (biology); Distributed computing; Algorithm; Artificial intelligence; Mathematics; Engineering","score_opus":0.02110450445621641,"score_gpt":0.2508811779073411,"score_spread":0.22977667345112468,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070960202","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013338684,0.00027885212,0.9838374,0.00036834515,0.000061001563,0.00004427791,0.000033259286,0.00013680881,0.0019013117],"genre_scores_gemma":[0.58170205,0.00046870313,0.4131698,0.00024007921,0.00012425307,0.0002650731,0.00013812765,0.00006312665,0.0038288687],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999516,0.00022775303,0.000024456212,0.00006084513,0.00013420524,0.00003681547],"domain_scores_gemma":[0.998755,0.0008692688,0.000114977454,0.000056477238,0.00016146447,0.000042837273],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012713512,0.0006986407,0.0010951213,0.00074647955,0.0004639171,0.00084095146,0.0010181037,0.0011853233,0.0012574465],"category_scores_gemma":[0.0039518545,0.00047244332,0.00044792646,0.0007306694,0.0008115275,0.0007072409,0.000753055,0.0009266435,0.00018863784],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024509358,0.000015667061,0.00026643745,0.000028074,0.000013119752,0.00001829741,0.000014698751,0.98403144,0.00026084244,0.0051968126,0.00027405014,0.009856158],"study_design_scores_gemma":[0.000005262582,0.000004249755,0.000031965526,0.0000013057057,0.0000011935778,0.0000023118764,0.0000012613608,0.99843377,0.000042061907,0.0014056348,0.00006974811,0.0000012076565],"about_ca_topic_score_codex":0.0078099417,"about_ca_topic_score_gemma":0.003193149,"teacher_disagreement_score":0.0078099417,"about_ca_system_score_codex":0.00083845266,"about_ca_system_score_gemma":0.00099904,"threshold_uncertainty_score":0.015528977},"labels":[],"label_agreement":null},{"id":"W2072069604","doi":"10.1109/saner.2015.7081820","title":"JCHARMING: A bug reproduction approach using crash traces and directed model checking","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Crash; Java; Computer science; Field (mathematics); Software; Code (set theory); Model checking; Source code; Software bug; Software engineering; Data science; Programming language; Set (abstract data type)","score_opus":0.15163872696120653,"score_gpt":0.3069914423862612,"score_spread":0.15535271542505466,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2072069604","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020805007,0.000100501915,0.95842016,0.00013440655,0.000033963563,0.00018602185,0.00019213834,0.0192702,0.00085761386],"genre_scores_gemma":[0.34466946,0.00019611913,0.650012,0.00017432818,0.000026651716,0.00026196954,0.000998335,0.0023863679,0.0012747806],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99682224,0.0008317754,0.00022162187,0.0005290034,0.0013246443,0.00027077523],"domain_scores_gemma":[0.98633134,0.005912222,0.0010764225,0.004586452,0.001807199,0.00028630916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003445585,0.0015829282,0.0009843741,0.003310342,0.0006821878,0.0015285828,0.0035948493,0.0013467047,0.0018107418],"category_scores_gemma":[0.013599532,0.0010217961,0.0023159005,0.0010227535,0.0016364518,0.0023097368,0.0029445705,0.0019422616,0.0004191232],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005973398,0.000861964,0.030066485,0.0010884121,0.00062212546,0.0015470017,0.0014641389,0.47661555,0.06399577,0.044196412,0.007026851,0.37191796],"study_design_scores_gemma":[0.00006756888,0.00014320397,0.0008252034,0.000044478675,0.00010983903,0.00027396364,0.00006968131,0.95844537,0.024648689,0.01226374,0.0030524086,0.000055780776],"about_ca_topic_score_codex":0.008204089,"about_ca_topic_score_gemma":0.008871829,"teacher_disagreement_score":0.008204089,"about_ca_system_score_codex":0.0010169123,"about_ca_system_score_gemma":0.0031149373,"threshold_uncertainty_score":0.018222213},"labels":[],"label_agreement":null},{"id":"W2072077040","doi":"","title":"JST: An Automatic Test Generation Tool for Industrial Java Applications with Strings","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Java; Symbolic execution; Scalability; Suite; String (physics); Programming language; Test suite; Pathfinder; Key (lock); Software engineering; Test case; Database; Operating system; Software; Machine learning","score_opus":0.04805541970660984,"score_gpt":0.26503095846209584,"score_spread":0.216975538755486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2072077040","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03967051,0.00013325455,0.8843664,0.00013932525,0.00006159604,0.00017634951,0.00078978535,0.07177604,0.0028868092],"genre_scores_gemma":[0.40031388,0.00016531807,0.58748215,0.00017960482,0.00004838993,0.0004684376,0.0031876366,0.0056731277,0.002481446],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9984162,0.0004560816,0.00015424429,0.00021185545,0.0006545633,0.00010704577],"domain_scores_gemma":[0.996797,0.0020865095,0.00030027976,0.00032654268,0.00041388406,0.00007573719],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001392144,0.0010665405,0.0005005126,0.0017728431,0.00037109983,0.0010035139,0.001473597,0.0008038514,0.0061871056],"category_scores_gemma":[0.0067065153,0.00046782632,0.00075506663,0.00085540663,0.00080522185,0.0010674136,0.0008884358,0.0006732977,0.0014855993],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011372163,0.00048191426,0.015883695,0.0012779769,0.00023158602,0.0020628956,0.00051325146,0.17347237,0.12934762,0.02354371,0.047442097,0.6046056],"study_design_scores_gemma":[0.00026796435,0.00034865283,0.002235166,0.00012703355,0.00005787437,0.0010058525,0.000061954684,0.83909476,0.12688352,0.010737282,0.019100673,0.00007936603],"about_ca_topic_score_codex":0.001540171,"about_ca_topic_score_gemma":0.0009929016,"teacher_disagreement_score":0.0061871056,"about_ca_system_score_codex":0.00042008256,"about_ca_system_score_gemma":0.00094667333,"threshold_uncertainty_score":0.020697951},"labels":[],"label_agreement":null},{"id":"W2073900130","doi":"10.5555/2133429.2133588","title":"Engineering a scalable Boolean matching based on EDA SaaS 2.0","year":2010,"lang":"en","type":"article","venue":"International Conference on Computer Aided Design","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Software as a service; Computer science; Scalability; Cloud computing; Overhead (engineering); Software; Distributed computing; Algorithm; Software development; Operating system","score_opus":0.04149444232221811,"score_gpt":0.2732101998238203,"score_spread":0.23171575750160217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2073900130","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03212364,0.00011902739,0.9599595,0.00022812071,0.000051003768,0.00013961292,0.00008122755,0.002468936,0.0048290025],"genre_scores_gemma":[0.30707833,0.00010659792,0.6891203,0.0001650267,0.000020717835,0.00013167104,0.00019250734,0.00014127589,0.0030436725],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991165,0.00016445566,0.00007707139,0.00017987214,0.00034743504,0.0001147246],"domain_scores_gemma":[0.99905044,0.00031946503,0.000096658,0.00024131926,0.0002582182,0.00003386872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007038359,0.00044672991,0.00040842005,0.00064401753,0.00048254675,0.0010360152,0.0010595012,0.00052770297,0.004218587],"category_scores_gemma":[0.0025455456,0.00030187948,0.00055073365,0.0007662741,0.0005463913,0.0017710405,0.0009868603,0.0005691921,0.00068572594],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004899072,0.0002909394,0.0026764756,0.00041435641,0.00010619205,0.00034517274,0.00020499936,0.15392011,0.13016169,0.122347705,0.0070191203,0.58202326],"study_design_scores_gemma":[0.00009525797,0.00023244126,0.00040315863,0.00003279684,0.000050738836,0.0003030013,0.00006948476,0.8833759,0.06839728,0.031801812,0.015215069,0.00002313994],"about_ca_topic_score_codex":0.002451977,"about_ca_topic_score_gemma":0.0032673127,"teacher_disagreement_score":0.004218587,"about_ca_system_score_codex":0.0011642971,"about_ca_system_score_gemma":0.0015754258,"threshold_uncertainty_score":0.014112532},"labels":[],"label_agreement":null},{"id":"W2074184938","doi":"10.5539/cis.v4n2p17","title":"Necessities and Tactics of Tool-Based Software Testing","year":2011,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Software engineering; System integration testing; Key (lock); Software; Test strategy; Set (abstract data type); Resource (disambiguation); Software development; Acceptance testing; Software construction; Computer security; Operating system; Programming language","score_opus":0.04193299607930647,"score_gpt":0.2424153438950541,"score_spread":0.2004823478157476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2074184938","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12665294,0.009940084,0.51236147,0.048777573,0.0007886452,0.00059783633,0.00011414488,0.0010793759,0.29968792],"genre_scores_gemma":[0.78396577,0.0036289305,0.20158784,0.0030695559,0.00056718726,0.0006211252,0.000099127865,0.00021815009,0.006242185],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.94653696,0.031195482,0.0027523944,0.0023345994,0.015784884,0.0013957504],"domain_scores_gemma":[0.89196366,0.07229021,0.0056156646,0.016646411,0.009484549,0.003999564],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024651168,0.00062798266,0.00063581567,0.004793101,0.002465567,0.007823692,0.0026362436,0.003026321,0.0024341946],"category_scores_gemma":[0.065881036,0.0005371906,0.0005526153,0.0025265708,0.014434625,0.009668048,0.005137529,0.004980138,0.0009090941],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000098071214,0.00013982426,0.00781974,0.0002996845,0.000024818193,0.0009944803,0.0035646465,0.0015638673,0.0016947611,0.8916638,0.0030322287,0.08910416],"study_design_scores_gemma":[0.00011240902,0.00061259884,0.007642756,0.001519029,0.000068659625,0.007121006,0.0057649934,0.019165484,0.005502693,0.82764536,0.12468815,0.00015682462],"about_ca_topic_score_codex":0.0006331701,"about_ca_topic_score_gemma":0.0007207947,"teacher_disagreement_score":0.024651168,"about_ca_system_score_codex":0.0018091258,"about_ca_system_score_gemma":0.0024960379,"threshold_uncertainty_score":0.13036942},"labels":[],"label_agreement":null},{"id":"W2074487274","doi":"10.5555/2819009.2819250","title":"8th international workshop on search-based software testing (SBST 2015)","year":2015,"lang":"en","type":"article","venue":"International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Oracle; Software engineering; Search-based software engineering; Computer science; Software testing; Software; Construct (python library); Test (biology); Service (business); Set (abstract data type); Software construction; Software development; Systems engineering; Engineering; Programming language","score_opus":0.11723149425156179,"score_gpt":0.32581176964823766,"score_spread":0.20858027539667587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2074487274","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01481958,0.059752796,0.68563753,0.023190629,0.046811067,0.0011566348,0.002492819,0.0070960466,0.15904284],"genre_scores_gemma":[0.11754519,0.04532361,0.40489364,0.007395992,0.01349604,0.001503759,0.015736679,0.0048849615,0.38922018],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9923902,0.0023206642,0.0006090128,0.0012422783,0.0026712401,0.00076652976],"domain_scores_gemma":[0.9911678,0.002438744,0.000298346,0.0016513916,0.0031152414,0.0013285006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008410317,0.0018027056,0.0015993973,0.0028582243,0.0010064394,0.0060878512,0.003251006,0.0031513763,0.0583417],"category_scores_gemma":[0.013856714,0.0007891261,0.001975964,0.0022487117,0.0014095964,0.005821453,0.005210569,0.004393615,0.025941119],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034358673,0.00035799947,0.00094607903,0.0006198715,0.00010706462,0.0002926582,0.0005470712,0.0048615476,0.0058434284,0.042088557,0.31822687,0.6257652],"study_design_scores_gemma":[0.00006255149,0.00039104975,0.0013540164,0.00077962456,0.00006313871,0.00072866847,0.00030585992,0.012867217,0.0044417516,0.03908583,0.9398478,0.00007244345],"about_ca_topic_score_codex":0.0022535578,"about_ca_topic_score_gemma":0.0028188305,"teacher_disagreement_score":0.0583417,"about_ca_system_score_codex":0.002006159,"about_ca_system_score_gemma":0.0031771632,"threshold_uncertainty_score":0.19517243},"labels":[],"label_agreement":null},{"id":"W2077772687","doi":"10.1016/s0950-5849(03)00060-0","title":"UIO sequence based checking sequences for distributed test architectures","year":2003,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Observability; Controllability; Sequence (biology); Sequence diagram; Synchronization (alternating current); Computer science; Current (fluid); Control theory (sociology); Real-time computing; Engineering; Mathematics; Software; Channel (broadcasting); Artificial intelligence; Unified Modeling Language; Computer network; Control (management); Programming language; Electrical engineering","score_opus":0.01946076082708154,"score_gpt":0.2629249510792006,"score_spread":0.24346419025211907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077772687","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06976175,0.00022403218,0.91089,0.00020113627,0.00009999436,0.0002520469,0.00025281316,0.013467209,0.0048511024],"genre_scores_gemma":[0.6148876,0.00008351295,0.37835577,0.00022600435,0.000038103062,0.00031657808,0.00068129797,0.00094091415,0.004470135],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958683,0.001698701,0.00039005582,0.0004984443,0.0010810241,0.0004633642],"domain_scores_gemma":[0.97867155,0.011074486,0.0017712604,0.0050410554,0.002889823,0.0005518454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003961203,0.00092557137,0.000789293,0.0023236908,0.0008866181,0.0014275793,0.0022187764,0.001310758,0.007182269],"category_scores_gemma":[0.018180106,0.00069308013,0.0006282883,0.0009941359,0.0013784486,0.0028483502,0.0019449363,0.0011913457,0.0010344367],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005216333,0.00082451385,0.013418236,0.0006449293,0.00011911871,0.00091961684,0.0009980839,0.090986356,0.051952127,0.12530948,0.010665326,0.6989459],"study_design_scores_gemma":[0.00037770098,0.0009804086,0.0021259324,0.00027981188,0.0001301179,0.00055935484,0.0001838146,0.7548718,0.11961491,0.107396685,0.013365638,0.000113854556],"about_ca_topic_score_codex":0.002641561,"about_ca_topic_score_gemma":0.0047143335,"teacher_disagreement_score":0.007182269,"about_ca_system_score_codex":0.0011575966,"about_ca_system_score_gemma":0.0019965256,"threshold_uncertainty_score":0.02402711},"labels":[],"label_agreement":null},{"id":"W2080458911","doi":"10.1109/pst.2010.5593250","title":"A model-driven penetration test framework for Web applications","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Web application; Software engineering; Web application security; Security testing; Web service; World Wide Web; Web development; Security information and event management; Operating system; Cloud computing security; Cloud computing","score_opus":0.029238329952034868,"score_gpt":0.3038530720129196,"score_spread":0.27461474206088476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2080458911","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020102316,0.000082391154,0.9943809,0.000113264185,0.000012792788,0.00019633536,0.000051371477,0.0023880277,0.0007648009],"genre_scores_gemma":[0.095306635,0.00030784277,0.9009992,0.000121570105,0.000024608556,0.00084925594,0.0003298099,0.0006113057,0.0014497216],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99651104,0.0014203984,0.00026987927,0.0002823031,0.0012826382,0.00023370565],"domain_scores_gemma":[0.9967698,0.0016174932,0.000307879,0.00052975403,0.0006662597,0.000108870896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045307647,0.0019564833,0.0010813593,0.0019034132,0.0006091714,0.0018344895,0.0030965966,0.0018698865,0.0022091854],"category_scores_gemma":[0.007399269,0.0010781283,0.002443281,0.00066882564,0.0014642522,0.0021103874,0.0017377514,0.0024688272,0.0005662533],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001384864,0.00043554968,0.0019360526,0.000572373,0.00021723377,0.001016125,0.00059615535,0.683406,0.020698667,0.15728708,0.0050371634,0.12865913],"study_design_scores_gemma":[0.000038337104,0.00013333642,0.00023783254,0.000103148544,0.000042459236,0.00032066804,0.000040041876,0.95554256,0.005545431,0.026931873,0.011013672,0.00005060426],"about_ca_topic_score_codex":0.008628542,"about_ca_topic_score_gemma":0.006942362,"teacher_disagreement_score":0.008628542,"about_ca_system_score_codex":0.001439289,"about_ca_system_score_gemma":0.002689062,"threshold_uncertainty_score":0.023961246},"labels":[],"label_agreement":null},{"id":"W2080670657","doi":"10.1002/spe.686","title":"Fast dynamic casting","year":2005,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lockheed Martin (Canada)","funders":"","keywords":"Pointer (user interface); C dynamic memory allocation; Computer science; Modulo; Integer (computer science); Class (philosophy); Arithmetic; Predictability; Base (topology); Theoretical computer science; Algorithm; Programming language; Discrete mathematics; Mathematics; Artificial intelligence; Memory management","score_opus":0.014209689347191417,"score_gpt":0.3035970456289335,"score_spread":0.2893873562817421,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2080670657","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015695177,0.00025050473,0.9438696,0.0002453662,0.00029671847,0.00034628567,0.00030565914,0.019345097,0.019645475],"genre_scores_gemma":[0.33125502,0.0004971081,0.62006134,0.00042138153,0.0001515357,0.0006241308,0.0011029036,0.0072826776,0.038603812],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963744,0.00054797553,0.0002781023,0.00042036222,0.0018146319,0.00056445587],"domain_scores_gemma":[0.99245983,0.0015126392,0.0003344224,0.00454462,0.0009272898,0.00022107767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002489152,0.001086455,0.0013778107,0.0014714047,0.001669324,0.003308318,0.0027562827,0.0015629526,0.021205643],"category_scores_gemma":[0.0081204055,0.0011484327,0.0010529311,0.0010774406,0.0021508632,0.0039353645,0.0076898276,0.0030822868,0.0054733977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001138235,0.00028916865,0.0024152643,0.0004366612,0.00009396531,0.0007350207,0.001363035,0.023650248,0.062364694,0.3893758,0.03066379,0.48747408],"study_design_scores_gemma":[0.00028819797,0.00029977623,0.0009837297,0.0002672247,0.00011996225,0.0011473271,0.00023988273,0.20399165,0.18875995,0.20122871,0.40241423,0.00025930535],"about_ca_topic_score_codex":0.0014123282,"about_ca_topic_score_gemma":0.0015636137,"teacher_disagreement_score":0.021205643,"about_ca_system_score_codex":0.0011508579,"about_ca_system_score_gemma":0.0016698971,"threshold_uncertainty_score":0.0709399},"labels":[],"label_agreement":null},{"id":"W2080932040","doi":"10.1145/1356058.1356065","title":"Near-optimal instruction selection on dags","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute of Steel Construction; National Science Foundation","keywords":"Computer science; Selection (genetic algorithm); Directed acyclic graph; Key (lock); Decomposition; Code (set theory); Tree (set theory); Theoretical computer science; Parallel computing; Algorithm; Mathematics; Programming language; Artificial intelligence","score_opus":0.025523420680057217,"score_gpt":0.24694817995428955,"score_spread":0.22142475927423233,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2080932040","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22694963,0.0008853698,0.76002795,0.00037191887,0.000056169025,0.00011171056,0.00028296091,0.005764407,0.005549892],"genre_scores_gemma":[0.6025336,0.0003286242,0.39259088,0.00017554474,0.000023481738,0.00011045747,0.0007199587,0.0003675786,0.0031498936],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99945134,0.0001666544,0.000038191523,0.00009127519,0.00014111197,0.0001113744],"domain_scores_gemma":[0.9988557,0.0007243238,0.00010351349,0.00012776264,0.00014911032,0.000039647166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047156843,0.00050890347,0.00056820427,0.0008803131,0.00037143464,0.00047355608,0.00068513147,0.00029127256,0.001850005],"category_scores_gemma":[0.002213739,0.0002305929,0.00032601654,0.00097053894,0.00045041795,0.00091819273,0.00054674817,0.0004360799,0.00043899554],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056559755,0.00015038945,0.0031925295,0.0003338391,0.000032734133,0.0003692588,0.00018444365,0.31091934,0.08248139,0.028990252,0.0086120535,0.56416816],"study_design_scores_gemma":[0.000078551115,0.0001670009,0.00062931963,0.000021474027,0.00002137864,0.0001571681,0.000083282896,0.89610565,0.035793904,0.061351668,0.0055708233,0.00001980107],"about_ca_topic_score_codex":0.0017766157,"about_ca_topic_score_gemma":0.0037122385,"teacher_disagreement_score":0.001850005,"about_ca_system_score_codex":0.00065581326,"about_ca_system_score_gemma":0.0010968043,"threshold_uncertainty_score":0.0061888695},"labels":[],"label_agreement":null},{"id":"W2083965123","doi":"10.1007/s10009-012-0240-3","title":"Model-based testing of software and systems: recent advances and challenges","year":2012,"lang":"en","type":"article","venue":"International Journal on Software Tools for Technology Transfer","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Software testing; Variety (cybernetics); System integration testing; Software engineering; Automation; Software; Test strategy; Model-based testing; Software performance testing; Software system; Data science; Software construction; Test case; Artificial intelligence; Machine learning; Programming language; Engineering","score_opus":0.09062809914528827,"score_gpt":0.30661703236558707,"score_spread":0.2159889332202988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2083965123","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04658733,0.29085517,0.6293575,0.014148121,0.000845782,0.00011755365,0.00015972696,0.0019390037,0.015989827],"genre_scores_gemma":[0.624386,0.15649721,0.20943874,0.0021462673,0.0020827646,0.0002176875,0.0007312492,0.0005967135,0.0039032924],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98997086,0.003906097,0.00045877259,0.0009270263,0.004453023,0.00028432315],"domain_scores_gemma":[0.9530092,0.03660307,0.0018952739,0.0036949618,0.004075221,0.00072229607],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010157286,0.0012406199,0.0025939508,0.0025116762,0.00033576696,0.0036023003,0.005114042,0.0025387541,0.002421341],"category_scores_gemma":[0.022252526,0.00063782313,0.0010747935,0.0030833054,0.0030104234,0.00788141,0.0023957426,0.002447246,0.000642589],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026874582,0.00044219458,0.006715571,0.0024048341,0.00019366064,0.00013417295,0.00027526641,0.03469002,0.0056594345,0.071145855,0.005469539,0.8726008],"study_design_scores_gemma":[0.00016477746,0.0011014426,0.006150268,0.002221374,0.00030938085,0.001981988,0.00077852176,0.6018795,0.011224488,0.2944967,0.07953287,0.00015868241],"about_ca_topic_score_codex":0.0015150171,"about_ca_topic_score_gemma":0.001341657,"teacher_disagreement_score":0.010157286,"about_ca_system_score_codex":0.001471828,"about_ca_system_score_gemma":0.0016661953,"threshold_uncertainty_score":0.053717494},"labels":[],"label_agreement":null},{"id":"W2084172672","doi":"10.1016/j.tcs.2011.07.010","title":"Hardness results for covering arrays avoiding forbidden edges and error-locating arrays","year":2011,"lang":"en","type":"article","venue":"Theoretical Computer Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Division of Electrical, Communications and Cyber Systems; Natural Sciences and Engineering Research Council of Canada","keywords":"Clique; Enhanced Data Rates for GSM Evolution; Binary number; Reduction (mathematics); Time complexity; Combinatorics; Hardness of approximation; Computer science; Computational complexity theory; Discrete mathematics; Mathematics; Algorithm; Approximation algorithm; Arithmetic; Artificial intelligence; Geometry","score_opus":0.057304972214434674,"score_gpt":0.28066317984675326,"score_spread":0.2233582076323186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084172672","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19810046,0.0016772925,0.7458919,0.005617522,0.00027496027,0.00023212368,0.0015612667,0.0024204375,0.04422404],"genre_scores_gemma":[0.8245463,0.0016194723,0.15579426,0.0015862546,0.0005881679,0.00036766758,0.0022865136,0.000874069,0.012337328],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9958068,0.0010707257,0.00026728833,0.00082470046,0.0013143187,0.00071602035],"domain_scores_gemma":[0.95496976,0.03756806,0.0016910356,0.0039148913,0.0010918947,0.00076444534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018146867,0.0018852152,0.0020251772,0.0016911896,0.0021420035,0.0035966071,0.004386523,0.0032305385,0.010437593],"category_scores_gemma":[0.024233095,0.0019292837,0.0030739477,0.002784229,0.003942846,0.01232188,0.004251009,0.00631158,0.001185517],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019430239,0.0007621286,0.008253856,0.0026012661,0.0005354344,0.0015572278,0.0021638903,0.26083288,0.017024191,0.5427304,0.035598606,0.12599717],"study_design_scores_gemma":[0.00022612353,0.00013424952,0.0016869693,0.000119492855,0.0002553319,0.00079448894,0.00037283514,0.18402515,0.006385953,0.7997652,0.0061513023,0.000082926956],"about_ca_topic_score_codex":0.0023123831,"about_ca_topic_score_gemma":0.0022701437,"teacher_disagreement_score":0.010437593,"about_ca_system_score_codex":0.0014024002,"about_ca_system_score_gemma":0.001305575,"threshold_uncertainty_score":0.034917176},"labels":[],"label_agreement":null},{"id":"W2084669353","doi":"10.1109/icpc.2013.6613851","title":"Towards generating human-oriented summaries of unit test cases","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Unit testing; Computer science; Java; USable; Test (biology); Code coverage; Code (set theory); Programming language; Test case; Unit (ring theory); Software engineering; Software; Machine learning; World Wide Web; Set (abstract data type)","score_opus":0.042413954737019616,"score_gpt":0.2900342707541397,"score_spread":0.24762031601712006,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084669353","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03959853,0.00039235057,0.94367415,0.00045436138,0.000081040045,0.0008254619,0.0011912832,0.011611966,0.0021708228],"genre_scores_gemma":[0.14412794,0.00032024283,0.847566,0.00015236191,0.00007690697,0.00061084016,0.0043778038,0.0015399226,0.001228042],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99253684,0.0037317227,0.0006367176,0.00089644786,0.0020061738,0.0001920504],"domain_scores_gemma":[0.93019116,0.04015008,0.006965859,0.009405071,0.012272259,0.0010156044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007027773,0.0019909353,0.0011532552,0.0040731095,0.00054442335,0.003662405,0.0023303728,0.0012256952,0.0030989808],"category_scores_gemma":[0.06931178,0.0007352179,0.0008894386,0.0018930157,0.0007160634,0.0021028535,0.0014250388,0.0013323454,0.0014642432],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009810169,0.0008622351,0.012143397,0.001680482,0.000222779,0.0019019396,0.0041362583,0.07892464,0.0312923,0.02303238,0.021882378,0.8229402],"study_design_scores_gemma":[0.0004418218,0.0017607185,0.006097278,0.00079034653,0.000325259,0.00144354,0.0018683086,0.7766512,0.07981342,0.05901055,0.071548745,0.00024882192],"about_ca_topic_score_codex":0.0017580697,"about_ca_topic_score_gemma":0.0029909285,"teacher_disagreement_score":0.007027773,"about_ca_system_score_codex":0.00092012907,"about_ca_system_score_gemma":0.002114217,"threshold_uncertainty_score":0.037166893},"labels":[],"label_agreement":null},{"id":"W2086037832","doi":"10.1016/j.cor.2007.01.013","title":"Detecting buffer overflow via automatic test input data generation","year":2007,"lang":"en","type":"article","venue":"Computers & Operations Research","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Buffer overflow; Computer science; Genetic programming; Process (computing); Fitness function; Software; Genetic algorithm; Buffer (optical fiber); Function (biology); Data mining; Real-time computing; Distributed computing; Machine learning; Operating system","score_opus":0.17966519198828013,"score_gpt":0.4136846375907641,"score_spread":0.234019445602484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2086037832","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3302644,0.00074959814,0.62134194,0.00049640785,0.00019389317,0.00021674047,0.0005966244,0.043721896,0.002418497],"genre_scores_gemma":[0.88014203,0.00006358393,0.11826635,0.000206311,0.000033265133,0.00009320109,0.0002581902,0.00030264605,0.00063445984],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966961,0.0011987877,0.00024239364,0.0005420427,0.0010673217,0.00025334532],"domain_scores_gemma":[0.97709274,0.0156078795,0.0023626864,0.0024045852,0.0020665657,0.00046551993],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015144179,0.0015811109,0.0011157102,0.0030844088,0.00037791883,0.0013346468,0.0017847363,0.0014615768,0.0026846319],"category_scores_gemma":[0.015283416,0.0005360753,0.00043407796,0.0011512456,0.000647406,0.0016817902,0.0013253626,0.00093737536,0.0005879287],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004392502,0.0006680907,0.04039326,0.00058615353,0.0002311239,0.0017512484,0.00047047535,0.03671739,0.19070815,0.0067419326,0.010342831,0.7069968],"study_design_scores_gemma":[0.00034528173,0.00084506586,0.008494874,0.00008977498,0.00017074813,0.0016869496,0.00007897764,0.69520164,0.2804653,0.00930668,0.0032164957,0.00009817508],"about_ca_topic_score_codex":0.0009974047,"about_ca_topic_score_gemma":0.0012046128,"teacher_disagreement_score":0.0030844088,"about_ca_system_score_codex":0.0007214992,"about_ca_system_score_gemma":0.0009531358,"threshold_uncertainty_score":0.00898093},"labels":[],"label_agreement":null},{"id":"W2086038725","doi":"10.1134/s0361768812040019","title":"FSM-based testing from user defined faults adapted to incremental and mutation testing","year":2012,"lang":"en","type":"article","venue":"Programming and Computer Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Test suite; Computer science; Model-based testing; Finite-state machine; Code coverage; Test case; Suite; Fault coverage; Polynomial; State (computer science); Algorithm; Programming language; Mathematics; Software; Engineering; Machine learning","score_opus":0.039164504726991324,"score_gpt":0.25790353807168404,"score_spread":0.21873903334469272,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2086038725","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05202641,0.00016622296,0.9445213,0.00012760688,0.000023083836,0.00011446977,0.000070011076,0.0017385735,0.0012123492],"genre_scores_gemma":[0.6145592,0.00012405758,0.38325644,0.00012316332,0.00003316028,0.00027361902,0.00046454053,0.00025376215,0.00091210124],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975097,0.0009706316,0.00014360793,0.00022659333,0.000988586,0.00016076825],"domain_scores_gemma":[0.99303037,0.00506802,0.00039138817,0.0009204357,0.00047736752,0.00011244996],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016884354,0.0007120333,0.0006619908,0.0012981014,0.00029491523,0.0005854526,0.0014073357,0.001014166,0.0010940962],"category_scores_gemma":[0.010200308,0.0003140042,0.0010464875,0.0009575745,0.0012653791,0.0010665003,0.00093327824,0.00097452936,0.0002222404],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028701298,0.00027796003,0.0031280841,0.00036760862,0.000104414896,0.00090089807,0.0002578808,0.66071445,0.041930567,0.044961277,0.0015386473,0.24553111],"study_design_scores_gemma":[0.00003708857,0.00014994093,0.00049438875,0.000019220479,0.00003132752,0.0002323693,0.000013324688,0.959386,0.016049355,0.022455879,0.001117436,0.000013708801],"about_ca_topic_score_codex":0.0014450916,"about_ca_topic_score_gemma":0.0012999536,"teacher_disagreement_score":0.0016884354,"about_ca_system_score_codex":0.0009926175,"about_ca_system_score_gemma":0.0009102017,"threshold_uncertainty_score":0.008929431},"labels":[],"label_agreement":null},{"id":"W2087097337","doi":"10.1145/1022494.1022539","title":"Relevant empirical testing research","year":2004,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Software testing; Empirical research; Test strategy; Computer science; Software engineering; Position paper; Software; Software performance testing; Data science; Systems engineering; Engineering; Software construction; Software development; World Wide Web","score_opus":0.1418802265110641,"score_gpt":0.360298768827243,"score_spread":0.21841854231617888,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2087097337","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15248236,0.25898933,0.05748237,0.123103864,0.0040234798,0.0014091198,0.007845428,0.0004412024,0.39422292],"genre_scores_gemma":[0.80573654,0.122263245,0.02416198,0.021512067,0.003680479,0.0012238411,0.0090126,0.0005748136,0.011834385],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9447487,0.030224562,0.0039454745,0.0049444693,0.01447959,0.0016571195],"domain_scores_gemma":[0.3908071,0.5199423,0.023959788,0.024461376,0.03644056,0.0043890146],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.043215863,0.0013036943,0.001524726,0.008815325,0.0023207765,0.0073999134,0.004079916,0.0031052642,0.034956284],"category_scores_gemma":[0.2884843,0.0007368627,0.0009837591,0.015926423,0.006483055,0.009159607,0.0032440186,0.0044322773,0.0055449903],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005510229,0.0027528005,0.14182699,0.012471938,0.00070420466,0.0010017599,0.00599492,0.002168892,0.00081411365,0.2744958,0.07058222,0.4866354],"study_design_scores_gemma":[0.0002948529,0.0009553932,0.18016227,0.04431692,0.0009583928,0.0030238365,0.022912765,0.0057944516,0.0036130513,0.27165288,0.46609446,0.00022076366],"about_ca_topic_score_codex":0.0054374645,"about_ca_topic_score_gemma":0.00433207,"teacher_disagreement_score":0.043215863,"about_ca_system_score_codex":0.0051902942,"about_ca_system_score_gemma":0.0068785865,"threshold_uncertainty_score":0.22855002},"labels":[],"label_agreement":null},{"id":"W2088296131","doi":"10.1016/j.scico.2006.04.009","title":"Special issue on Source code analysis and manipulation","year":2006,"lang":"en","type":"article","venue":"Science of Computer Programming","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Program slicing; Source code; Slicing; Program comprehension; Transformation (genetics); Static program analysis; Code (set theory); Program transformation; Restructuring; Programming language; Program analysis; Software engineering; World Wide Web; Software; Software development; Software system; Set (abstract data type)","score_opus":0.014948064418239062,"score_gpt":0.265467116510335,"score_spread":0.25051905209209596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2088296131","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010352342,0.04427118,0.025630765,0.023454992,0.7937981,0.00027568344,0.0009234367,0.0015945621,0.10901608],"genre_scores_gemma":[0.003818344,0.028496085,0.0059325635,0.0054772533,0.6412729,0.00022307567,0.0022721058,0.0016157096,0.310892],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99702007,0.00045016178,0.0002634062,0.0006164656,0.0014213815,0.00022854487],"domain_scores_gemma":[0.98908293,0.003009498,0.0005631548,0.0016830786,0.004059877,0.0016014463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027083973,0.0033275082,0.0041996706,0.008018537,0.0022845382,0.006987062,0.0029758317,0.003992754,0.13404925],"category_scores_gemma":[0.008312091,0.0011448392,0.0024316218,0.0058128317,0.0014936499,0.00631742,0.0032865861,0.0040040645,0.06852749],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000040741354,0.000070154616,0.00016760378,0.00037909605,0.000024826742,0.00008138498,0.000020374184,0.00016184825,0.00044841593,0.0032735744,0.92381483,0.07151719],"study_design_scores_gemma":[0.000015297084,0.00006146075,0.0005993116,0.00018724041,0.000032066677,0.00029789846,0.000021221478,0.00046317704,0.00029554582,0.006701652,0.99130756,0.00001753918],"about_ca_topic_score_codex":0.000877174,"about_ca_topic_score_gemma":0.002624053,"teacher_disagreement_score":0.13404925,"about_ca_system_score_codex":0.0015437595,"about_ca_system_score_gemma":0.0020857668,"threshold_uncertainty_score":0.44843942},"labels":[],"label_agreement":null},{"id":"W2088946319","doi":"10.1145/1984642.1984645","title":"Branching and merging","year":2011,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Branching (polymer chemistry); Computer science; Software; Branching process; Software engineering; Data science; Programming language; Mathematics","score_opus":0.038325349564623884,"score_gpt":0.23013589781330276,"score_spread":0.19181054824867888,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2088946319","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46886337,0.0046366034,0.42645785,0.005360795,0.00024078836,0.0005049538,0.00027512718,0.0022551941,0.09140528],"genre_scores_gemma":[0.8687194,0.001280846,0.12069606,0.00071118487,0.00007492657,0.00016189247,0.00023922598,0.0004538837,0.0076625617],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9728037,0.01385674,0.0019140828,0.0030927414,0.007051073,0.001281751],"domain_scores_gemma":[0.897254,0.06471636,0.01113662,0.017374799,0.0074139205,0.0021043783],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014665976,0.0005514359,0.00048932934,0.0029225606,0.0025125819,0.004292989,0.0017799193,0.0015280982,0.0048403377],"category_scores_gemma":[0.09720692,0.000654342,0.0005709311,0.0031102502,0.0054528215,0.013137731,0.004985015,0.002004575,0.0008158012],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035804624,0.00015918403,0.051596485,0.0005955251,0.00006443612,0.00082992675,0.06872138,0.0019625388,0.010375278,0.26454353,0.0056332154,0.5951604],"study_design_scores_gemma":[0.00011212369,0.00070517574,0.053708974,0.0017745403,0.00025193667,0.0071109394,0.025084011,0.01910403,0.023691548,0.48300353,0.38511592,0.0003372417],"about_ca_topic_score_codex":0.002282511,"about_ca_topic_score_gemma":0.0023294352,"teacher_disagreement_score":0.014665976,"about_ca_system_score_codex":0.0014293016,"about_ca_system_score_gemma":0.0022231466,"threshold_uncertainty_score":0.077562034},"labels":[],"label_agreement":null},{"id":"W2090566883","doi":"10.1109/icsm.2010.5609695","title":"Exploring the impact of context sensitivity on blended analysis","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"National Science Foundation","keywords":"Computer science; Scalability; Context (archaeology); Pruning; Call graph; Source code; Sensitivity (control systems); Empirical research; Context model; Object (grammar); Theoretical computer science; Data mining; Artificial intelligence; Database; Programming language; Mathematics; Engineering","score_opus":0.07496066508202304,"score_gpt":0.3072169454212202,"score_spread":0.23225628033919715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2090566883","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6532128,0.0004659074,0.34092125,0.00025455927,0.000029118655,0.0001977319,0.000057040874,0.0021238243,0.0027378302],"genre_scores_gemma":[0.89515716,0.00007300716,0.10410503,0.00007950545,0.000012211744,0.000045482393,0.00004621992,0.0002609734,0.00022045875],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98694104,0.0067418707,0.00066246645,0.0016450788,0.003260563,0.0007489942],"domain_scores_gemma":[0.8970022,0.082010634,0.0041614817,0.011838447,0.003939769,0.0010474258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007752712,0.0013383924,0.0008202892,0.0017902002,0.0011607761,0.0028565424,0.0019188036,0.0013343628,0.00095041934],"category_scores_gemma":[0.07820838,0.0008383832,0.00076898286,0.0011218725,0.0019429862,0.0060631055,0.0037380082,0.0022005683,0.00023355991],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026567967,0.0011802029,0.09793635,0.00054690224,0.00057968515,0.001012493,0.0048492234,0.19826661,0.1522041,0.017301975,0.00077124394,0.52269435],"study_design_scores_gemma":[0.0000877711,0.0011693854,0.019159444,0.00015558445,0.0004436265,0.0009490395,0.0009924517,0.83539116,0.11994905,0.018193882,0.0033412692,0.00016730693],"about_ca_topic_score_codex":0.0017583033,"about_ca_topic_score_gemma":0.0023961277,"teacher_disagreement_score":0.007752712,"about_ca_system_score_codex":0.0007543115,"about_ca_system_score_gemma":0.0012912325,"threshold_uncertainty_score":0.041000783},"labels":[],"label_agreement":null},{"id":"W2092198121","doi":"10.1016/j.jss.2012.12.051","title":"A survey of software testing practices in Canada","year":2013,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":168,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Software testing; Test (biology); Survey research; Software; Engineering; Computer science; Psychology; Applied psychology","score_opus":0.06050471719038154,"score_gpt":0.2706308278705806,"score_spread":0.2101261106801991,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2092198121","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.871035,0.04697299,0.0018853814,0.011469603,0.000224502,0.00033414783,0.011677781,0.0003420442,0.056058533],"genre_scores_gemma":[0.95147693,0.027100062,0.0022171382,0.0024015128,0.000033228764,0.000076984914,0.0026882335,0.000120569755,0.013885232],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99289024,0.00042340043,0.0005505293,0.00066371396,0.004125809,0.0013462729],"domain_scores_gemma":[0.9423838,0.008392444,0.0054996493,0.0006736644,0.03447647,0.008573928],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0025790378,0.00038224508,0.00052043935,0.009695053,0.004579158,0.003094968,0.0024629952,0.0009267799,0.005292805],"category_scores_gemma":[0.017149104,0.0005227926,0.0004342439,0.024284597,0.0016228473,0.0010291389,0.0012971256,0.0009334587,0.0005867861],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004400791,0.00022220242,0.68431586,0.0012570317,0.00011767309,0.001021689,0.011598965,0.00082632795,0.0019291135,0.0039974814,0.03048984,0.26378372],"study_design_scores_gemma":[0.000021586151,0.0000999816,0.909713,0.0007945314,0.000051258965,0.000525748,0.010735795,0.00061512255,0.00054561876,0.00017999229,0.07664217,0.000075268385],"about_ca_topic_score_codex":0.989943,"about_ca_topic_score_gemma":0.9936674,"teacher_disagreement_score":0.99742097,"about_ca_system_score_codex":0.066141024,"about_ca_system_score_gemma":0.13182211,"threshold_uncertainty_score":0.47988898},"labels":[],"label_agreement":null},{"id":"W2092245922","doi":"10.1016/j.infsof.2013.03.004","title":"Graphical user interface (GUI) testing: Systematic mapping and repository","year":2013,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":167,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Graphical user interface testing; Graphical user interface; Computer science; User interface; Software; Interface (matter); Information retrieval; Data mining; Software engineering; User interface design; Programming language; Operating system","score_opus":0.010876144624941233,"score_gpt":0.21512095863092082,"score_spread":0.20424481400597957,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2092245922","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07386839,0.0027537171,0.9007135,0.0008015806,0.00006903624,0.001651856,0.0011195226,0.007948518,0.011073878],"genre_scores_gemma":[0.33791453,0.00197447,0.6522949,0.00027431315,0.00003436696,0.0009039903,0.0021228087,0.0011295963,0.0033510334],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.96947485,0.01570686,0.0024771246,0.0019832093,0.009592635,0.00076520693],"domain_scores_gemma":[0.9064861,0.043386262,0.006799906,0.01790901,0.024636144,0.0007826102],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01209981,0.0015487913,0.0011805862,0.012406146,0.0014438996,0.003815389,0.0038486999,0.0015441352,0.002708847],"category_scores_gemma":[0.077458024,0.00096189487,0.0012771459,0.006200252,0.0017910858,0.006727997,0.004184421,0.0015092382,0.001396955],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017023759,0.00074189063,0.028086137,0.0023134963,0.00014599013,0.00051699585,0.0037524346,0.005628964,0.010129589,0.0130532915,0.004786159,0.93067485],"study_design_scores_gemma":[0.00039209917,0.004606061,0.092382945,0.009605181,0.0018782552,0.014825426,0.019367771,0.29893258,0.27757517,0.13269994,0.14701599,0.00071866694],"about_ca_topic_score_codex":0.004841406,"about_ca_topic_score_gemma":0.008874613,"teacher_disagreement_score":0.012406146,"about_ca_system_score_codex":0.0013404308,"about_ca_system_score_gemma":0.007884588,"threshold_uncertainty_score":0.06399071},"labels":[],"label_agreement":null},{"id":"W2095773889","doi":"10.1109/icsme.2014.85","title":"A Web Service Test Generator","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"The King's University","funders":"","keywords":"Computer science; Web service; XML; XML validation; Programming language; Information retrieval; Database; Generator (circuit theory); Efficient XML Interchange; World Wide Web","score_opus":0.01603739895257633,"score_gpt":0.23558729255487795,"score_spread":0.21954989360230162,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095773889","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017873153,0.0001317514,0.8830227,0.00027306585,0.00019324361,0.0016485166,0.0028417814,0.0824835,0.011532188],"genre_scores_gemma":[0.25146386,0.00022904795,0.7049112,0.0005323432,0.00009591279,0.0028778655,0.014154484,0.009770774,0.015964512],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99833065,0.00042819377,0.00011805914,0.00030496513,0.0006816647,0.00013649827],"domain_scores_gemma":[0.9972398,0.0012871923,0.00011546171,0.0004997638,0.000723159,0.00013457761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015444638,0.0013256832,0.0007633261,0.0018244487,0.00046997916,0.0012612549,0.001822136,0.0011097271,0.021077348],"category_scores_gemma":[0.0064481082,0.0006561705,0.00097068376,0.00093501154,0.0005342028,0.0010675804,0.0013047104,0.0011546039,0.007919233],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012477682,0.00093905354,0.008285131,0.0011966238,0.00020492815,0.0048121926,0.00058836304,0.091135964,0.098809056,0.0586555,0.10859478,0.6255306],"study_design_scores_gemma":[0.0004196248,0.0004468133,0.0021904442,0.00015085544,0.00006978163,0.0022571925,0.00013606342,0.73564434,0.14638232,0.029120151,0.08309626,0.00008625317],"about_ca_topic_score_codex":0.0012426759,"about_ca_topic_score_gemma":0.00077263353,"teacher_disagreement_score":0.021077348,"about_ca_system_score_codex":0.00073231943,"about_ca_system_score_gemma":0.001632339,"threshold_uncertainty_score":0.070510685},"labels":[],"label_agreement":null},{"id":"W2095794856","doi":"10.1007/978-1-4471-0719-4_32","title":"On Built-in-Test Classes for Object-Oriented and Component-Based Information Systems","year":2001,"lang":"en","type":"book-chapter","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Testability; Class (philosophy); Object-oriented programming; Software; Software construction; Software development; Component-based software engineering; Programming language; Computer architecture; Software engineering; Engineering; Reliability engineering","score_opus":0.021670964125821573,"score_gpt":0.24998184568840734,"score_spread":0.22831088156258578,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095794856","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003395543,0.014228883,0.90536815,0.0009952104,0.0007139093,0.00016365541,0.00016581145,0.005100017,0.06986877],"genre_scores_gemma":[0.06768893,0.023702176,0.7781641,0.0009136946,0.000928413,0.00037376027,0.00097556674,0.004520044,0.12273336],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988011,0.0002152525,0.00010930938,0.00016668199,0.00061128446,0.00009637048],"domain_scores_gemma":[0.99724966,0.001889084,0.000109680914,0.00044577196,0.0002418755,0.00006385897],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012271461,0.002132087,0.0011879219,0.0017747511,0.00082984776,0.0030515755,0.0027888732,0.0019329627,0.012580245],"category_scores_gemma":[0.0046653547,0.0014051495,0.0011062683,0.0027276613,0.0026072853,0.006143252,0.001563013,0.0045823436,0.0051937713],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054704054,0.00015625687,0.0003647824,0.00064667425,0.000026783893,0.00025834094,0.0007658559,0.00775495,0.0036165467,0.3335263,0.034963634,0.6178651],"study_design_scores_gemma":[0.000060693303,0.0001168567,0.0010833979,0.0008212192,0.00008367165,0.0017011707,0.00012342623,0.035409123,0.008999261,0.5784166,0.37310538,0.00007922266],"about_ca_topic_score_codex":0.0022803196,"about_ca_topic_score_gemma":0.0029214767,"teacher_disagreement_score":0.012580245,"about_ca_system_score_codex":0.0011466563,"about_ca_system_score_gemma":0.0008122787,"threshold_uncertainty_score":0.04208511},"labels":[],"label_agreement":null},{"id":"W2095902286","doi":"10.1109/itng.2009.306","title":"Lessons Learned from a Survey of Web Applications Testing","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; World Wide Web; Software testing; The Internet; Government (linguistics); Web testing; Web page; Web application security; Web application; Data science; Web development; Software; Operating system","score_opus":0.18010079232447723,"score_gpt":0.3500422421938346,"score_spread":0.16994144986935739,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095902286","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23753761,0.42087924,0.13942191,0.14332573,0.0023437778,0.00034535467,0.0021207149,0.0016000821,0.05242565],"genre_scores_gemma":[0.7068091,0.20883809,0.052013416,0.019449199,0.002331429,0.00025039908,0.0029566756,0.0006411417,0.0067106327],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.97770613,0.010183631,0.0025940351,0.0024315591,0.0062303203,0.000854332],"domain_scores_gemma":[0.73386616,0.21489947,0.006686245,0.0051388373,0.037295084,0.002114159],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024004929,0.000778281,0.00094512955,0.0077925315,0.0007890659,0.0035318078,0.0016650598,0.0016301506,0.002395003],"category_scores_gemma":[0.098709635,0.00047746295,0.0005437334,0.00862488,0.0017767011,0.01081146,0.0012160294,0.0021180483,0.0008200583],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012891357,0.00020612376,0.065945074,0.0026635656,0.00007920985,0.00044272898,0.00392834,0.0014879883,0.00092360994,0.0122689055,0.037363674,0.87456197],"study_design_scores_gemma":[0.00007947427,0.0016754625,0.24387279,0.012328852,0.00035857284,0.008587937,0.02732282,0.01495534,0.006751561,0.04963976,0.63412654,0.00030091172],"about_ca_topic_score_codex":0.0047223344,"about_ca_topic_score_gemma":0.006686188,"teacher_disagreement_score":0.024004929,"about_ca_system_score_codex":0.001659372,"about_ca_system_score_gemma":0.0015227107,"threshold_uncertainty_score":0.1269517},"labels":[],"label_agreement":null},{"id":"W2095974519","doi":"10.1109/tse.2010.32","title":"Assessing, Comparing, and Combining State Machine-Based Testing and Structural Testing: A Series of Experiments","year":2010,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Test strategy; Non-regression testing; Software performance testing; White-box testing; Machine learning; Reliability engineering; Model-based testing; Software testing; Risk-based testing; Software; Series (stratigraphy); Code coverage; State (computer science); Keyword-driven testing; Software reliability testing; Data mining; Test case; Software quality; Algorithm; Software system; Software development; Engineering; Software construction","score_opus":0.034421962619943035,"score_gpt":0.2701052537879602,"score_spread":0.23568329116801714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095974519","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97860765,0.00020046812,0.017885275,0.000085902466,0.000052065774,0.0013315949,0.00028820123,0.0002922184,0.0012566929],"genre_scores_gemma":[0.9555268,0.00020497966,0.03990018,0.00011580654,0.000038802027,0.0023806794,0.0006231075,0.00008808241,0.0011214742],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9917631,0.004088154,0.00090368197,0.0011800597,0.0015791826,0.00048579942],"domain_scores_gemma":[0.9114702,0.07438717,0.0038898138,0.0058213347,0.003445126,0.0009864353],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007070511,0.0018822305,0.001360801,0.0012413356,0.0004689528,0.0008518606,0.002129822,0.0017532419,0.0021390193],"category_scores_gemma":[0.03658542,0.00067412434,0.0011200706,0.00089196773,0.001152671,0.0020088542,0.0011708707,0.0012317771,0.00035883562],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.05284751,0.0768852,0.02951685,0.0034656643,0.0016389025,0.0007389088,0.003403259,0.2880326,0.23557474,0.0056206253,0.002859672,0.29941615],"study_design_scores_gemma":[0.0085663,0.25125974,0.03511932,0.00023322666,0.0013912476,0.0006388863,0.001437558,0.42150342,0.26246235,0.008947087,0.0079782745,0.00046254008],"about_ca_topic_score_codex":0.0016843845,"about_ca_topic_score_gemma":0.001675944,"teacher_disagreement_score":0.007070511,"about_ca_system_score_codex":0.00105869,"about_ca_system_score_gemma":0.0010838535,"threshold_uncertainty_score":0.037392914},"labels":[],"label_agreement":null},{"id":"W2096226167","doi":"10.1109/issre.2003.1251027","title":"A comprehensive and systematic methodology for client-server class integration testing","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Class (philosophy); Context (archaeology); Client–server model; Data mining; Server; Database; Artificial intelligence; World Wide Web","score_opus":0.16833415468862511,"score_gpt":0.3493839636123504,"score_spread":0.1810498089237253,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096226167","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00029611448,0.00011004921,0.99843353,0.00007510467,0.000008987545,0.00034015466,0.000024715622,0.0003856176,0.0003256657],"genre_scores_gemma":[0.00714553,0.00021322859,0.991297,0.000067152265,0.000013904118,0.00072191487,0.00010555661,0.000114608265,0.00032107544],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.96026564,0.016606519,0.004381946,0.002521997,0.015551555,0.00067232095],"domain_scores_gemma":[0.9714139,0.0103311045,0.0020304928,0.007486818,0.008181455,0.0005561516],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0251837,0.002728123,0.0025441598,0.0077872314,0.0019999298,0.0053569963,0.005894163,0.0026301313,0.002847305],"category_scores_gemma":[0.038671847,0.0021530073,0.002508931,0.0036322537,0.005079355,0.0060081733,0.0049395883,0.0046786326,0.0018811122],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007626913,0.00091041456,0.004061597,0.0037732741,0.00034223983,0.0010068441,0.005256298,0.031555828,0.024885293,0.25019228,0.0073201815,0.6706195],"study_design_scores_gemma":[0.0002954885,0.0018347264,0.005289475,0.0084669115,0.00070575345,0.010075788,0.0043638796,0.27914828,0.071253985,0.37888494,0.23882683,0.0008539796],"about_ca_topic_score_codex":0.001923743,"about_ca_topic_score_gemma":0.002937301,"teacher_disagreement_score":0.0251837,"about_ca_system_score_codex":0.0022165214,"about_ca_system_score_gemma":0.010665801,"threshold_uncertainty_score":0.13318568},"labels":[],"label_agreement":null},{"id":"W2096282282","doi":"10.1109/scam.2005.27","title":"Transforming Embedded Java Code into Custom Tags","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Java; Java applet; Programming language; Java API for XML-based RPC; Real time Java; Business logic; Java annotation; Source code; Code (set theory); Web page; Operating system; World Wide Web; Set (abstract data type)","score_opus":0.011676163790382232,"score_gpt":0.2534006728177438,"score_spread":0.2417245090273616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096282282","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1013303,0.000091775066,0.8762085,0.00014061062,0.00021052387,0.00021736781,0.00011357897,0.013907754,0.0077796546],"genre_scores_gemma":[0.4964724,0.0003215199,0.47931537,0.00029249254,0.00003966592,0.00016459517,0.000637959,0.00619739,0.016558621],"study_design_codex":"bench_or_experimental","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992335,0.00007933382,0.00007765786,0.00014397701,0.00037443367,0.00009116596],"domain_scores_gemma":[0.99670166,0.0009785456,0.00037447395,0.0011981487,0.0006738815,0.00007319629],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070806564,0.00050108926,0.00023786967,0.0005282029,0.00023795203,0.0010593544,0.00088311237,0.0005715104,0.0024507956],"category_scores_gemma":[0.0047536744,0.00047391056,0.00041730126,0.0004402043,0.0007432264,0.0014327187,0.0008727816,0.0011651769,0.0011437971],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047210715,0.0005670129,0.0064281435,0.0005022584,0.00010359025,0.0023306885,0.0016897022,0.021975027,0.51393557,0.049030177,0.0058833044,0.39708236],"study_design_scores_gemma":[0.00010832839,0.00028393613,0.0031318506,0.0001291326,0.00009334485,0.0018901324,0.00028946256,0.11784203,0.76858705,0.023662632,0.08388975,0.00009235339],"about_ca_topic_score_codex":0.00073586835,"about_ca_topic_score_gemma":0.0008517218,"teacher_disagreement_score":0.0024507956,"about_ca_system_score_codex":0.00034189995,"about_ca_system_score_gemma":0.0005615786,"threshold_uncertainty_score":0.008198738},"labels":[],"label_agreement":null},{"id":"W2096830010","doi":"10.1109/apsec.2003.1254396","title":"Automated test generation from object-oriented specifications of real-time reactive systems","year":2004,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Oracle; Object-oriented programming; Real-time computing; Domain (mathematical analysis); System testing; Reliability engineering; Software engineering; Programming language; Engineering","score_opus":0.03609809509497559,"score_gpt":0.2620086390616272,"score_spread":0.2259105439666516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096830010","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08329146,0.00013269775,0.9068503,0.0001692964,0.00004486738,0.00043370164,0.00037275182,0.0069053303,0.0017995997],"genre_scores_gemma":[0.40113562,0.00025013986,0.59252506,0.00020316704,0.000036595622,0.00078307895,0.0023986932,0.0012362555,0.0014314231],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99342585,0.0030525734,0.00047255377,0.00033587596,0.0024214513,0.00029171404],"domain_scores_gemma":[0.96079457,0.029531518,0.0027372614,0.0027881735,0.0037433733,0.00040505268],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044674617,0.0012342723,0.00074563274,0.0021378146,0.0003308822,0.0014062586,0.001653181,0.0013154447,0.0020819067],"category_scores_gemma":[0.027003672,0.00059813866,0.0009239589,0.0010639583,0.0011208993,0.0011676752,0.0011096838,0.00075960177,0.00058232027],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012782447,0.00088743074,0.012773452,0.0013592569,0.00022215326,0.0036881373,0.0023792721,0.3064481,0.0869737,0.040376827,0.006836374,0.53677696],"study_design_scores_gemma":[0.00045797936,0.000524768,0.0019081164,0.00015191945,0.000095004594,0.00080788415,0.00034485632,0.84863555,0.11364541,0.02469875,0.008641112,0.00008867334],"about_ca_topic_score_codex":0.001684103,"about_ca_topic_score_gemma":0.0021160305,"teacher_disagreement_score":0.0044674617,"about_ca_system_score_codex":0.0006321848,"about_ca_system_score_gemma":0.0012763349,"threshold_uncertainty_score":0.023626447},"labels":[],"label_agreement":null},{"id":"W2096914257","doi":"10.1109/test.1997.639710","title":"Supervisors for testing non-deterministically specified systems","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Bell (Canada); University of Waterloo","funders":"","keywords":"Supervisor; Computer science; Reliability engineering; System testing; Determinism; Software engineering; Engineering","score_opus":0.11584181985453662,"score_gpt":0.2614392815401561,"score_spread":0.1455974616856195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096914257","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012220965,0.000354079,0.98043025,0.00011162894,0.000040123574,0.00015797398,0.00010180752,0.004853918,0.0017292899],"genre_scores_gemma":[0.41167706,0.00046313455,0.5825462,0.00015148474,0.00006508446,0.0007590877,0.00062884437,0.00043759696,0.0032715946],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9970805,0.0012079461,0.00018341596,0.0002958506,0.0010872401,0.00014500959],"domain_scores_gemma":[0.99224114,0.005399137,0.000766978,0.0009164476,0.00046794643,0.00020840723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021667082,0.0008271892,0.0006929259,0.00091988273,0.00039999594,0.0010874837,0.0011337072,0.0007107503,0.0026009947],"category_scores_gemma":[0.0067063216,0.00058750465,0.00058812875,0.00042142242,0.0012926599,0.0009803347,0.0009573591,0.0020437506,0.0006982367],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011346779,0.00050213636,0.0064446414,0.0014953315,0.0001382426,0.0015841062,0.0016497817,0.19797122,0.08493925,0.37494636,0.008138022,0.32105622],"study_design_scores_gemma":[0.00021870116,0.0005842563,0.0010569491,0.00024486828,0.00008816511,0.00066644134,0.00017560198,0.80677265,0.053191796,0.11621556,0.02072479,0.000060284714],"about_ca_topic_score_codex":0.0007903429,"about_ca_topic_score_gemma":0.0010921645,"teacher_disagreement_score":0.0026009947,"about_ca_system_score_codex":0.00060605985,"about_ca_system_score_gemma":0.0013184764,"threshold_uncertainty_score":0.011458755},"labels":[],"label_agreement":null},{"id":"W2097180485","doi":"10.1109/test.1993.470702","title":"On the design for testability of communication software","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Testability; Computer science; Software engineering; Software; Protocol (science); Communications protocol; Software design; Software testing; Software development; Programming language; Reliability engineering; Engineering; Operating system; Medicine","score_opus":0.10298193677832075,"score_gpt":0.27574761034728895,"score_spread":0.17276567356896821,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2097180485","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013405188,0.0011236251,0.97820204,0.0010072753,0.00006130398,0.00021954533,0.0000287972,0.00053941796,0.0054128137],"genre_scores_gemma":[0.36216024,0.0023200244,0.63114387,0.0006243273,0.00028317506,0.0011058525,0.00021653967,0.00043420796,0.0017117484],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9660067,0.01886872,0.002502601,0.0019357597,0.00944632,0.0012398576],"domain_scores_gemma":[0.84374887,0.13090214,0.006682872,0.009622689,0.008105667,0.0009376756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021285072,0.0018756507,0.0013673981,0.0025921047,0.0010510695,0.0037272521,0.002340422,0.0026763016,0.0023260089],"category_scores_gemma":[0.101139285,0.0009373249,0.0014202062,0.0014677619,0.007476758,0.006685016,0.0025488834,0.003444647,0.00061581045],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004048359,0.0003342782,0.004629564,0.0015021121,0.00017850472,0.0008810728,0.0016696305,0.088141374,0.014784297,0.5890618,0.0023859378,0.29602665],"study_design_scores_gemma":[0.00041910278,0.0012207793,0.001748855,0.0011727995,0.00029788044,0.0016401856,0.00044214856,0.2691486,0.04827462,0.64573413,0.029751202,0.0001496973],"about_ca_topic_score_codex":0.0014442124,"about_ca_topic_score_gemma":0.00075847306,"teacher_disagreement_score":0.021285072,"about_ca_system_score_codex":0.0018953914,"about_ca_system_score_gemma":0.0022639418,"threshold_uncertainty_score":0.112567544},"labels":[],"label_agreement":null},{"id":"W2098174190","doi":"10.5555/2337223.2337449","title":"Locating features in dynamically configured avionics software","year":2012,"lang":"en","type":"article","venue":"International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Avionics software; Computer science; Avionics; Software construction; Program comprehension; Software system; Software engineering; Software; Static program analysis; Software development; Verification and validation; Software framework; Software sizing; Component-based software engineering; Embedded system; Programming language; Engineering","score_opus":0.02324148763626,"score_gpt":0.2753308825634161,"score_spread":0.2520893949271561,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098174190","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.64977086,0.00032224026,0.3374081,0.000067475514,0.000018746226,0.00005556331,0.00010291367,0.01093707,0.0013170068],"genre_scores_gemma":[0.9347119,0.000075487995,0.064198665,0.000016236721,0.0000034004427,0.00002216399,0.00009848017,0.00033809958,0.0005356601],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99953115,0.0000856235,0.00002692986,0.00010734268,0.00018993867,0.000058988582],"domain_scores_gemma":[0.9981237,0.00084004673,0.00037548275,0.00036951428,0.0002334242,0.000057719488],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00030177762,0.0005448683,0.00034679804,0.0012260217,0.0004152782,0.0007318273,0.0010289522,0.0007786619,0.001002489],"category_scores_gemma":[0.0028488406,0.00040662035,0.00030886705,0.00058878807,0.0006447382,0.0012920229,0.0007948933,0.00048321602,0.00023623894],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079863943,0.00017290149,0.026253257,0.00033391453,0.000058002628,0.0033364715,0.0019445,0.104667954,0.56448454,0.0087566115,0.0013271744,0.2878661],"study_design_scores_gemma":[0.000070542344,0.0007123216,0.017071169,0.000115617906,0.000110932335,0.002334287,0.00039305477,0.6371282,0.32290962,0.013335856,0.005699326,0.00011904572],"about_ca_topic_score_codex":0.0015421492,"about_ca_topic_score_gemma":0.0017754675,"teacher_disagreement_score":0.0015421492,"about_ca_system_score_codex":0.0004726787,"about_ca_system_score_gemma":0.00042968232,"threshold_uncertainty_score":0.003429532},"labels":[],"label_agreement":null},{"id":"W2098272565","doi":"10.1504/ijict.2007.013274","title":"Software testing: a graph theoretic approach","year":2007,"lang":"en","type":"article","venue":"International Journal of Information and Communication Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Control flow graph; Software testing; White-box testing; Software; Path (computing); Graph; Software performance testing; Control flow; Software engineering; Theoretical computer science; Software development; Programming language; Software construction","score_opus":0.014552573438138433,"score_gpt":0.2665073864543589,"score_spread":0.25195481301622047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098272565","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004419739,0.0010380301,0.978954,0.0014483193,0.000059609283,0.000068427165,0.00006124711,0.00020469805,0.013745923],"genre_scores_gemma":[0.4271326,0.0049022464,0.5562634,0.0011313148,0.0005605152,0.0003638771,0.00029206433,0.00026416557,0.009089786],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988105,0.0005032304,0.000043469816,0.00015406904,0.0004107525,0.00007795965],"domain_scores_gemma":[0.99574393,0.003371849,0.00016109426,0.00027654113,0.0003508921,0.00009571646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010549352,0.0010938325,0.0009581996,0.0032765546,0.00076927204,0.002088458,0.0019094542,0.0014666484,0.0044587473],"category_scores_gemma":[0.0049971677,0.00061005715,0.0011004222,0.0022180523,0.0039982228,0.0039562793,0.0011119504,0.0019335744,0.00060126313],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016873428,0.000050875893,0.0003308744,0.00017918702,0.000029370469,0.00012892578,0.00011792201,0.076810256,0.0009163824,0.8654976,0.0021947701,0.053726956],"study_design_scores_gemma":[0.000011789664,0.000029669416,0.00013636482,0.000050875595,0.00001696302,0.000120653756,0.00005403689,0.14479764,0.00050423556,0.8475941,0.006671406,0.0000123238415],"about_ca_topic_score_codex":0.0028704132,"about_ca_topic_score_gemma":0.0019271335,"teacher_disagreement_score":0.0044587473,"about_ca_system_score_codex":0.0016867965,"about_ca_system_score_gemma":0.000952384,"threshold_uncertainty_score":0.014916003},"labels":[],"label_agreement":null},{"id":"W2098280085","doi":"10.1145/2593069.2602976","title":"Safety Evaluation of Automotive Electronics Using Virtual Prototypes","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Infineon Technologies (Canada)","funders":"","keywords":"Dependability; Automotive electronics; Automotive industry; Electronics; Automotive engineering; Computer science; Functional safety; Reliability engineering; Manufacturing engineering; Engineering; Embedded system; Electrical engineering","score_opus":0.035692471732468395,"score_gpt":0.3091718003874681,"score_spread":0.2734793286549997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098280085","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97034407,0.00021656265,0.025916599,0.000038316823,0.000060853356,0.000085387655,0.00007585219,0.00036616344,0.00289622],"genre_scores_gemma":[0.99322575,0.00006103646,0.005896678,0.000006568386,0.0000060195107,0.000029065583,0.00009088268,0.000025268444,0.0006586176],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99858737,0.00056474254,0.00006746142,0.0001133657,0.0005381415,0.00012884283],"domain_scores_gemma":[0.995615,0.0018974507,0.00040961528,0.0006337304,0.00124828,0.00019598173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015312079,0.00077751116,0.00033032053,0.0010238425,0.00027197122,0.0007404199,0.0010181682,0.00078357273,0.0024527772],"category_scores_gemma":[0.0058635785,0.00025645766,0.0004169263,0.00027803282,0.00065469346,0.0007491072,0.0008599939,0.00029248558,0.0003044809],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007043531,0.0029457544,0.021413833,0.0010771238,0.00023809115,0.0011737224,0.0018887001,0.4066749,0.28822768,0.0051845126,0.0022458846,0.26188633],"study_design_scores_gemma":[0.0003879672,0.024956957,0.026041824,0.000127798,0.00027464566,0.0009164858,0.0011211737,0.6823881,0.2531851,0.004173177,0.006296818,0.00012993228],"about_ca_topic_score_codex":0.00041600937,"about_ca_topic_score_gemma":0.00031743676,"teacher_disagreement_score":0.0024527772,"about_ca_system_score_codex":0.000430062,"about_ca_system_score_gemma":0.00023849941,"threshold_uncertainty_score":0.008205414},"labels":[],"label_agreement":null},{"id":"W2098647975","doi":"10.1109/compsac.2006.98","title":"A Multi-Agent Framework for Testing Distributed Systems","year":2006,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Software performance testing; Distributed computing; Software architecture; System integration testing; Software reliability testing; Task (project management); Process (computing); Software system; Software engineering; Software; Software construction; Systems engineering; Operating system; Engineering","score_opus":0.0935818281614061,"score_gpt":0.32300081754675025,"score_spread":0.22941898938534416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098647975","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001974168,0.000336486,0.9935616,0.00025999814,0.000040753443,0.00016154048,0.000021651214,0.0006383011,0.0030054692],"genre_scores_gemma":[0.11163562,0.0004236227,0.8849934,0.00008896167,0.000050179773,0.00041321004,0.000075227756,0.00008488904,0.0022349283],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971777,0.001423781,0.00015884479,0.0002420408,0.0008722927,0.00012541114],"domain_scores_gemma":[0.9981818,0.0009372306,0.00015418942,0.0002965216,0.00026852643,0.0001616597],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032572593,0.0011156502,0.00088925654,0.00085918134,0.0010214078,0.0022328077,0.00296642,0.0019213677,0.002214778],"category_scores_gemma":[0.0043578926,0.0005327224,0.0010297376,0.0007468705,0.002069291,0.0018979135,0.001685854,0.0023707163,0.0005564385],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010026487,0.00019326614,0.0008770272,0.00039766944,0.00012322821,0.00092952844,0.00058063574,0.32281786,0.007104097,0.5486345,0.003746373,0.11449552],"study_design_scores_gemma":[0.00008691443,0.00010099764,0.00019128951,0.00009934301,0.000036565703,0.0002737215,0.00007441129,0.82789016,0.0024219884,0.13840084,0.03038408,0.000039762694],"about_ca_topic_score_codex":0.0055095274,"about_ca_topic_score_gemma":0.005414791,"teacher_disagreement_score":0.0055095274,"about_ca_system_score_codex":0.0011279505,"about_ca_system_score_gemma":0.0018588329,"threshold_uncertainty_score":0.01722622},"labels":[],"label_agreement":null},{"id":"W2098933948","doi":"10.1109/icstw.2010.35","title":"Supporting Test-Driven Development of Graphical User Interfaces Using Agile Interaction Design","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Graphical user interface; Graphical user interface testing; Agile software development; Scripting language; Human–computer interaction; Test (biology); User interface; Test script; Test-driven development; Software engineering; Test Management Approach; Fidelity; Programming language; Test case; User interface design; Software development; Software","score_opus":0.05133768931905014,"score_gpt":0.33723895008860305,"score_spread":0.2859012607695529,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098933948","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0120111955,0.00007501572,0.984006,0.00011389396,0.000017743761,0.00018097082,0.000018318606,0.0019779615,0.0015989754],"genre_scores_gemma":[0.1496532,0.00023065008,0.8474986,0.00011718701,0.000026476853,0.0005164819,0.00013892895,0.00070226996,0.001116217],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98826236,0.0064819823,0.00083717494,0.0006364428,0.0031311715,0.0006509611],"domain_scores_gemma":[0.95430344,0.026885638,0.0025753083,0.0092417,0.0056936196,0.0013001603],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01161993,0.0014686112,0.0006809754,0.0011716193,0.0005008784,0.0027562522,0.0025453786,0.0013090464,0.0020163138],"category_scores_gemma":[0.04026716,0.0010024474,0.0006259639,0.00043292542,0.0012276828,0.0028982826,0.0033742448,0.0023124502,0.0010449034],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047804505,0.000995434,0.0092640305,0.001186237,0.0002294913,0.0023413491,0.005459746,0.055098265,0.10949776,0.040509865,0.0060435436,0.76889616],"study_design_scores_gemma":[0.00080979144,0.0028102077,0.0053040176,0.0010442224,0.00024012591,0.005326407,0.0016309697,0.62620646,0.20535883,0.062325843,0.088596456,0.0003466612],"about_ca_topic_score_codex":0.00056262437,"about_ca_topic_score_gemma":0.00063390855,"teacher_disagreement_score":0.01161993,"about_ca_system_score_codex":0.000445371,"about_ca_system_score_gemma":0.0014178448,"threshold_uncertainty_score":0.061452746},"labels":[],"label_agreement":null},{"id":"W2099441126","doi":"10.1109/tse.2003.1214327","title":"General test result checking with log file analysis","year":2003,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":95,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Random testing; Formalism (music); Programming language; Unit testing; Operating system; Software; Test case","score_opus":0.011147187604348262,"score_gpt":0.21500120278407897,"score_spread":0.2038540151797307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099441126","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007896787,0.00007660389,0.98087364,0.000092103066,0.00001678504,0.0001476252,0.00012692307,0.00981705,0.00095251703],"genre_scores_gemma":[0.31692448,0.00016372801,0.67774165,0.00025091952,0.000082385974,0.0006486533,0.00097246474,0.0014334787,0.0017823555],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9863593,0.0045703594,0.0013133332,0.0015297352,0.005437585,0.00078964775],"domain_scores_gemma":[0.96384144,0.017055323,0.0031446638,0.012552184,0.0030804514,0.00032589884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005819703,0.0016131802,0.0012983641,0.0032776836,0.00071523216,0.0032212331,0.0040414254,0.0016840895,0.0051211435],"category_scores_gemma":[0.036806498,0.00080929877,0.001706109,0.00183454,0.0033134886,0.006775377,0.0033063265,0.0019516095,0.001431096],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018721181,0.0009540087,0.017678656,0.00185562,0.00030088646,0.0018022475,0.0010512437,0.111706346,0.061787598,0.18405148,0.014171024,0.6027688],"study_design_scores_gemma":[0.00025620218,0.0006021031,0.0023314634,0.00032876688,0.00019285691,0.002089937,0.00013428011,0.70148623,0.122601226,0.1508023,0.018990496,0.00018417792],"about_ca_topic_score_codex":0.0021805032,"about_ca_topic_score_gemma":0.0016057021,"teacher_disagreement_score":0.005819703,"about_ca_system_score_codex":0.0013185618,"about_ca_system_score_gemma":0.002989001,"threshold_uncertainty_score":0.030777931},"labels":[],"label_agreement":null},{"id":"W2099523200","doi":"10.1109/mutation.2006.5","title":"<i>ExMAn</i>: A Generic and Customizable Framework for Experimental Mutation Analysis","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Mutation testing; Computer science; Static analysis; Mutation; Software testing; Programming language; Software engineering; Quality assurance; Quality (philosophy); Software; Engineering","score_opus":0.015969552038483013,"score_gpt":0.27566231475820446,"score_spread":0.25969276271972147,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099523200","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030744926,0.000044939465,0.96145284,0.0000910681,0.0000522804,0.000335281,0.0008317375,0.031372625,0.0027446738],"genre_scores_gemma":[0.046278022,0.00008738019,0.94401,0.00018914609,0.00004789021,0.0013019647,0.0015494803,0.004823005,0.0017130848],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99588937,0.0010577891,0.0004920251,0.0007390935,0.0015460936,0.00027577006],"domain_scores_gemma":[0.98813593,0.0040151905,0.001151515,0.0051554837,0.0012896479,0.0002521664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0078802295,0.0015486571,0.0011255988,0.0026303201,0.000966973,0.0020399245,0.0040923837,0.0014995214,0.011356064],"category_scores_gemma":[0.017558431,0.0011093899,0.0013879383,0.0012988565,0.0020588278,0.002963057,0.003298048,0.00258809,0.0023388513],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00092471077,0.001270114,0.012874826,0.0015608668,0.0004144671,0.0010708823,0.00059532217,0.07458211,0.16004184,0.19383805,0.06339215,0.48943475],"study_design_scores_gemma":[0.00030874918,0.0007671059,0.00712319,0.00036021092,0.00017109756,0.002224536,0.00015525045,0.44776383,0.28035718,0.09639007,0.16394897,0.00042982918],"about_ca_topic_score_codex":0.0018011086,"about_ca_topic_score_gemma":0.0021330249,"teacher_disagreement_score":0.011356064,"about_ca_system_score_codex":0.0010720646,"about_ca_system_score_gemma":0.0018067823,"threshold_uncertainty_score":0.04167509},"labels":[],"label_agreement":null},{"id":"W2100147370","doi":"10.5381/jot.2009.8.3.a4","title":"Automated State-Based Unit Testing for Aspect-Oriented Programs: A Supporting Framework.","year":2009,"lang":"en","type":"article","venue":"The Journal of Object Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières; Innovation and Economic Development Trois Rivières","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Unit testing; State (computer science); Unit (ring theory); Aspect-oriented programming; Software engineering; Reliability engineering; Programming language; Engineering; Software; Mathematics","score_opus":0.029179515527060974,"score_gpt":0.3179984803389712,"score_spread":0.2888189648119102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100147370","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020038583,0.00017333863,0.994165,0.0001275984,0.00001627179,0.000105835454,0.000028988878,0.0023731252,0.0010060209],"genre_scores_gemma":[0.074759185,0.00031619373,0.9229924,0.000060647162,0.00004871747,0.00020609016,0.00022519426,0.0003453537,0.0010462075],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99792385,0.00050147105,0.00013692914,0.00025170142,0.001057069,0.00012892619],"domain_scores_gemma":[0.9956722,0.0022221955,0.00042816543,0.00091070094,0.00058243406,0.00018436332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003682177,0.0010367375,0.0008113818,0.0018258125,0.00080715894,0.0024963324,0.0026348599,0.0015100733,0.0029566179],"category_scores_gemma":[0.0077734385,0.00083916885,0.0015995587,0.00091910124,0.0022162653,0.00282832,0.0016753322,0.0016193998,0.00096985494],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018999672,0.00057841296,0.0038719447,0.0007625114,0.0001866238,0.0008491083,0.00074720534,0.11549853,0.029592594,0.33702913,0.0061332425,0.5045607],"study_design_scores_gemma":[0.0000599252,0.00014902336,0.0006466493,0.00023794024,0.00006761936,0.00057717424,0.00006616252,0.85918105,0.011850908,0.10692376,0.020187335,0.000052427193],"about_ca_topic_score_codex":0.0031732935,"about_ca_topic_score_gemma":0.0030850323,"teacher_disagreement_score":0.003682177,"about_ca_system_score_codex":0.0009996417,"about_ca_system_score_gemma":0.002149123,"threshold_uncertainty_score":0.019473493},"labels":[],"label_agreement":null},{"id":"W2100209048","doi":"","title":"Static And Dynamic Reverse Engineering Techniques for Java Software Systems","year":2019,"lang":"en","type":"book","venue":"Trepo - Institutional Repository of Tampere University","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":81,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Tekes; University of Victoria","keywords":"Reverse engineering; Computer science; Java; Software engineering; Programming language","score_opus":0.008589263300448567,"score_gpt":0.19626188704935338,"score_spread":0.1876726237489048,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100209048","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021517184,0.00017325483,0.99234337,0.00009354815,0.000028068753,0.000046782705,0.000036209738,0.0009100432,0.0042170356],"genre_scores_gemma":[0.09187728,0.00075145136,0.8964073,0.00015030388,0.000041982516,0.00017039718,0.00033638245,0.00087375834,0.0093911495],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9967781,0.00067366334,0.00029947417,0.00056183274,0.0014587739,0.00022814753],"domain_scores_gemma":[0.9970912,0.0012891022,0.0002393307,0.00083027745,0.0005049584,0.000045067907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020415704,0.00093136885,0.00048426757,0.0016941468,0.00076257694,0.0025835286,0.0016942177,0.00076571066,0.007954699],"category_scores_gemma":[0.006072253,0.0010727568,0.002130115,0.0012251188,0.0012497838,0.004002583,0.0027989286,0.002589964,0.0018577407],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009896105,0.00012895068,0.0009877925,0.0008137547,0.00007992761,0.0004917815,0.0014073594,0.040237214,0.035236638,0.41614196,0.0041307868,0.5002449],"study_design_scores_gemma":[0.00010049387,0.0001946661,0.00096112984,0.0005075107,0.00024891383,0.0013091736,0.00094950706,0.402169,0.099424,0.3021057,0.191918,0.00011198847],"about_ca_topic_score_codex":0.0023241064,"about_ca_topic_score_gemma":0.004144958,"teacher_disagreement_score":0.007954699,"about_ca_system_score_codex":0.0009186727,"about_ca_system_score_gemma":0.001477943,"threshold_uncertainty_score":0.02661109},"labels":[],"label_agreement":null},{"id":"W2100485841","doi":"10.1109/infcom.1994.337662","title":"Fault coverage analysis in respect to an FSM specification","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Fault coverage; Test suite; Computer science; Fault tree analysis; Finite-state machine; Fault (geology); Algorithm; Fault model; Class (philosophy); Suite; Code coverage; Minification; Automatic test pattern generation; State (computer science); Test case; Theoretical computer science; Programming language; Reliability engineering; Artificial intelligence; Software; Engineering; Machine learning; Electronic circuit","score_opus":0.04951585261241914,"score_gpt":0.2916565817219161,"score_spread":0.24214072910949697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100485841","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0526631,0.00016367153,0.9423747,0.00023651289,0.000011680729,0.00009351809,0.0003391631,0.0014141584,0.002703423],"genre_scores_gemma":[0.69626695,0.00033613454,0.29848555,0.00018312743,0.000052149662,0.00032173196,0.0018818317,0.00033273848,0.0021398966],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99756527,0.0006208769,0.00012615624,0.00035858664,0.0011002372,0.0002289153],"domain_scores_gemma":[0.9926974,0.0058649657,0.00040599055,0.0003419934,0.0006297965,0.000059826067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011718671,0.0007407657,0.00074918516,0.00199852,0.00058626704,0.0012037135,0.0006224885,0.000729714,0.0028968607],"category_scores_gemma":[0.009316105,0.000278466,0.0010156028,0.00085240783,0.001077066,0.0013836403,0.0006252352,0.00074560044,0.00041754797],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005082513,0.00013086345,0.0051159556,0.00049229397,0.00016886415,0.0010073808,0.00033567834,0.617234,0.05891929,0.10180101,0.0036553438,0.21063112],"study_design_scores_gemma":[0.000030727955,0.00015487167,0.002058577,0.00006271091,0.00006860798,0.00036992473,0.0000677558,0.8860568,0.04324504,0.06440314,0.0034551122,0.00002678801],"about_ca_topic_score_codex":0.0030811117,"about_ca_topic_score_gemma":0.002368915,"teacher_disagreement_score":0.0030811117,"about_ca_system_score_codex":0.0009786381,"about_ca_system_score_gemma":0.0013030322,"threshold_uncertainty_score":0.009691},"labels":[],"label_agreement":null},{"id":"W2100775847","doi":"10.1145/1321631.1321654","title":"Nighthawk","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Software testing; Software","score_opus":0.013701328256595972,"score_gpt":0.2658061073748248,"score_spread":0.2521047791182288,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100775847","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0104246875,0.0018263628,0.05906793,0.002589095,0.0019144972,0.00038850438,0.011251862,0.0431172,0.8694199],"genre_scores_gemma":[0.03767412,0.00087843084,0.032750443,0.0007985681,0.0002192532,0.000145494,0.0109123895,0.0087655885,0.90785563],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99887866,0.00012387763,0.00004343681,0.00034780713,0.0004948029,0.00011147035],"domain_scores_gemma":[0.99732035,0.00034821648,0.00008400648,0.0007827148,0.00095514324,0.00050948065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010512812,0.00063543103,0.0006002282,0.0014748683,0.0014334305,0.0026745128,0.0016937065,0.000961675,0.39300388],"category_scores_gemma":[0.0029591813,0.00053995807,0.00052796403,0.0011008652,0.00058395986,0.0024730947,0.002456718,0.001310232,0.19247685],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000667075,0.00015722707,0.0014236484,0.00026233506,0.000024425592,0.00025582378,0.0002720127,0.0008806045,0.005076385,0.021940319,0.56658787,0.4024522],"study_design_scores_gemma":[0.000054543154,0.000053947108,0.0009483438,0.000039975395,0.000007819338,0.00021114883,0.000072807656,0.0016630704,0.0027061636,0.0036417667,0.99057907,0.000021286323],"about_ca_topic_score_codex":0.005467602,"about_ca_topic_score_gemma":0.01083497,"teacher_disagreement_score":0.39300388,"about_ca_system_score_codex":0.0011374338,"about_ca_system_score_gemma":0.0016651764,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2100888697","doi":"10.1109/pccc.1988.10096","title":"Modelling of expressed system behavior towards characterization of the conformance testing problem","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Nondeterministic algorithm; Computer science; Set (abstract data type); Conformance testing; Implementation; Characterization (materials science); Programming language; Tree (set theory); Theoretical computer science; Algorithm; Artificial intelligence; Data mining; Mathematics; Standardization; Operating system","score_opus":0.05406417772731281,"score_gpt":0.23563388366252355,"score_spread":0.18156970593521074,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100888697","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005858085,0.000053222615,0.99232036,0.00009489052,0.00000834128,0.00005081003,0.000057085595,0.00034776176,0.0012094504],"genre_scores_gemma":[0.3151096,0.00049174635,0.6778562,0.00017007237,0.00004255846,0.0006049443,0.00077586505,0.00043190623,0.004517218],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99669015,0.0013843101,0.00023209234,0.00039433953,0.0010869985,0.00021203468],"domain_scores_gemma":[0.9956501,0.0025688964,0.00042247924,0.0007539785,0.0005144585,0.00009021984],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022294926,0.00096997136,0.0005907597,0.00096477,0.00044361854,0.0019057583,0.0021142885,0.0016227121,0.002256259],"category_scores_gemma":[0.008792492,0.0005350314,0.0011601517,0.0008831851,0.0021903224,0.0028534797,0.0010339638,0.0020906765,0.0006928266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015932288,0.00011000278,0.001234959,0.00024539768,0.000038745453,0.0006932474,0.0009700054,0.3770618,0.020865766,0.5629512,0.0014403297,0.034229252],"study_design_scores_gemma":[0.000019654006,0.000056100867,0.00015094399,0.000057069934,0.000023608007,0.00019100115,0.000053517324,0.82441485,0.00847411,0.16097318,0.0055695935,0.000016398077],"about_ca_topic_score_codex":0.0020001088,"about_ca_topic_score_gemma":0.0013089217,"teacher_disagreement_score":0.002256259,"about_ca_system_score_codex":0.0010477613,"about_ca_system_score_gemma":0.0010408454,"threshold_uncertainty_score":0.011790872},"labels":[],"label_agreement":null},{"id":"W2100942157","doi":"10.1109/csmr.2010.21","title":"Automating Coverage Metrics for Dynamic Web Applications","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Instrumentation (computer programming); Web application; The Internet; Code coverage; Database; Set (abstract data type); Software; Operating system; Programming language","score_opus":0.012044583795526614,"score_gpt":0.2872492392723478,"score_spread":0.27520465547682116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100942157","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19642925,0.0005984548,0.78785497,0.00013983098,0.000020759575,0.0003133674,0.000839836,0.011457225,0.0023463143],"genre_scores_gemma":[0.66635704,0.0002836762,0.3283852,0.000038253398,0.000028178272,0.00039859966,0.0029027518,0.0010382393,0.0005680787],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.993371,0.002064747,0.00066499104,0.00073236896,0.0027094258,0.00045735078],"domain_scores_gemma":[0.97067845,0.019066673,0.0032231165,0.0024316974,0.0041535315,0.0004465539],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003457474,0.0013464906,0.0014209307,0.009768836,0.000667428,0.0018548727,0.0012001455,0.00082854205,0.0012328865],"category_scores_gemma":[0.029320441,0.00062912836,0.00088042504,0.0031567928,0.0006875579,0.0018385332,0.0019445405,0.00076251075,0.0004362167],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027097316,0.0002771738,0.05505652,0.00058968196,0.00014946047,0.0005816046,0.0008344237,0.20970961,0.06373908,0.0103464,0.0036796106,0.65476537],"study_design_scores_gemma":[0.000026487001,0.00015247044,0.013741435,0.00006921244,0.000044504923,0.00035917474,0.00014484457,0.9405194,0.0322031,0.009646806,0.0030383186,0.000054357235],"about_ca_topic_score_codex":0.0044952333,"about_ca_topic_score_gemma":0.004510951,"teacher_disagreement_score":0.009768836,"about_ca_system_score_codex":0.001154152,"about_ca_system_score_gemma":0.0010373906,"threshold_uncertainty_score":0.018285096},"labels":[],"label_agreement":null},{"id":"W2102418068","doi":"10.1109/tse.2002.1049402","title":"Timed Wp-method: testing real-time systems","year":2002,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Nondeterministic algorithm; Correctness; Timed automaton; Automaton; System under test; Finite-state machine; Fault coverage; Real-time operating system; Test case; Real-time computing; Distributed computing; Algorithm; Theoretical computer science; Embedded system","score_opus":0.026720613989180138,"score_gpt":0.23424799388672066,"score_spread":0.20752737989754053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102418068","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017787261,0.000079057194,0.97820765,0.000040283998,0.000022459924,0.00016110128,0.00017809871,0.0026322675,0.00089181546],"genre_scores_gemma":[0.3384065,0.00013720409,0.6577503,0.00007711513,0.000024796906,0.0005239425,0.0007294917,0.000625727,0.0017248413],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99785846,0.0007516762,0.00017018586,0.00044345448,0.000670169,0.00010597851],"domain_scores_gemma":[0.9958978,0.0025394445,0.0003507762,0.00076986034,0.0003586235,0.00008337678],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011310381,0.0012176683,0.00047070326,0.0007330701,0.00017078125,0.0004943058,0.0021824485,0.00075739034,0.003019999],"category_scores_gemma":[0.0061264904,0.00033055682,0.00072343455,0.00063532987,0.00076387846,0.0010477833,0.000606149,0.0006283829,0.0004527835],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007433625,0.00036308254,0.010028492,0.0021642926,0.00033041986,0.0017609013,0.0005391372,0.32070854,0.13414524,0.039218828,0.004840839,0.48515695],"study_design_scores_gemma":[0.00018702217,0.0004940294,0.0019521584,0.00010910156,0.00010386623,0.0011929697,0.00009255286,0.8449922,0.118267946,0.024052521,0.008509327,0.00004621792],"about_ca_topic_score_codex":0.0013722988,"about_ca_topic_score_gemma":0.00087739347,"teacher_disagreement_score":0.003019999,"about_ca_system_score_codex":0.00038804614,"about_ca_system_score_gemma":0.0008638132,"threshold_uncertainty_score":0.010102868},"labels":[],"label_agreement":null},{"id":"W2102816978","doi":"10.1155/2013/420394","title":"Regression Test Reduction for Object-Oriented Software: A Control Call Graph Based Technique and Associated Tool","year":2013,"lang":"en","type":"article","venue":"ISRN Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Regression testing; Test suite; Computer science; Test case; Control flow graph; Call graph; Control flow; Test Management Approach; Reduction (mathematics); Graph; Test (biology); Keyword-driven testing; Software regression; Model-based testing; Software; Control (management); Data mining; Regression analysis; Programming language; Artificial intelligence; Machine learning; Theoretical computer science; Software system; Software development; Software quality; Software construction; Mathematics","score_opus":0.006770740737484236,"score_gpt":0.21378199578570348,"score_spread":0.20701125504821924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102816978","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021057757,0.00010018261,0.96932685,0.00008304479,0.00001182634,0.00012108394,0.000045821515,0.0075106514,0.0017427732],"genre_scores_gemma":[0.3663168,0.00023607114,0.62852585,0.00018120011,0.000039105555,0.00026894995,0.0004040425,0.0015400072,0.0024879118],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99831486,0.0004643438,0.00007381335,0.00018382422,0.00086058996,0.0001026535],"domain_scores_gemma":[0.9957539,0.0027871502,0.00039313588,0.0006383264,0.0003671884,0.000060307222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009810962,0.00085787696,0.00054196303,0.0023905106,0.00032235627,0.0005307533,0.0013338069,0.00058588776,0.0020086514],"category_scores_gemma":[0.0046865577,0.00029524186,0.00087861944,0.0009364028,0.0008894925,0.00084100914,0.00088883296,0.000993044,0.00046373074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034535333,0.0005543565,0.005421926,0.00046461602,0.0001040002,0.0016384146,0.00056665303,0.062406313,0.12505646,0.028157312,0.0036600374,0.7716245],"study_design_scores_gemma":[0.0001255387,0.00077774253,0.0070321877,0.0001453541,0.00023229467,0.0037916042,0.00015508097,0.7151583,0.22114964,0.027518395,0.023785722,0.00012809888],"about_ca_topic_score_codex":0.0012557212,"about_ca_topic_score_gemma":0.0011534737,"teacher_disagreement_score":0.0023905106,"about_ca_system_score_codex":0.00027621054,"about_ca_system_score_gemma":0.0005568046,"threshold_uncertainty_score":0.0067195892},"labels":[],"label_agreement":null},{"id":"W2103745340","doi":"10.1109/ase.2002.1115029","title":"Adding value to formal test oracles","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test (biology); Programming language; Code coverage; Value (mathematics); Test case; Test Management Approach; Keyword-driven testing; Process (computing); Formal verification; Test harness; Software engineering; Software; Theoretical computer science; Software development; Machine learning; Software construction","score_opus":0.03135776340264899,"score_gpt":0.25354025767666805,"score_spread":0.22218249427401907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2103745340","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0103379125,0.00065414305,0.9773064,0.0016189833,0.00025794207,0.0001145052,0.0001510774,0.0044054217,0.0051536113],"genre_scores_gemma":[0.3850959,0.00087830273,0.60729486,0.0013758391,0.0006231087,0.00042155894,0.00069404696,0.0014525703,0.0021638237],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.95532626,0.021529775,0.0034645547,0.0030709186,0.0143920835,0.0022164776],"domain_scores_gemma":[0.82119715,0.12322834,0.0059707044,0.035147425,0.012635583,0.0018207596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017820874,0.0022612412,0.0020760433,0.0044591855,0.0008745177,0.007902778,0.0039984165,0.0033674445,0.004179183],"category_scores_gemma":[0.15889965,0.0013725942,0.001986107,0.002491207,0.008383112,0.018279428,0.0066188686,0.0061056186,0.0011180589],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030900288,0.00016944826,0.0053101648,0.00079461414,0.00014028263,0.0005196958,0.0008590919,0.04481525,0.0061427625,0.67309636,0.004703493,0.26313978],"study_design_scores_gemma":[0.00014136663,0.00032281005,0.0007902375,0.0005043902,0.00020127365,0.00064702396,0.00018650646,0.20279542,0.014595532,0.7387735,0.040879864,0.00016204658],"about_ca_topic_score_codex":0.00093783403,"about_ca_topic_score_gemma":0.0009922397,"teacher_disagreement_score":0.017820874,"about_ca_system_score_codex":0.002413563,"about_ca_system_score_gemma":0.00217661,"threshold_uncertainty_score":0.094246924},"labels":[],"label_agreement":null},{"id":"W2103792020","doi":"10.1016/j.comnet.2008.11.003","title":"Overcoming controllability problems with fewest channels between testers","year":2008,"lang":"en","type":"article","venue":"Computer Networks","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Controllability; Computer science; Port (circuit theory); Sequence (biology); Set (abstract data type); Software deployment; Test (biology); Distributed computing; Mathematics; Engineering","score_opus":0.028572891636590967,"score_gpt":0.22464161361730314,"score_spread":0.19606872198071218,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2103792020","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17186737,0.00029325837,0.81941456,0.00053625857,0.0000921346,0.00018503249,0.000072209354,0.0055040354,0.002035213],"genre_scores_gemma":[0.8877092,0.000056470788,0.109992854,0.00019186221,0.00009462671,0.00015678389,0.00007399551,0.000346919,0.0013772637],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9825514,0.009176255,0.0009510575,0.0027018087,0.0025752664,0.0020441033],"domain_scores_gemma":[0.7596581,0.15041229,0.013700916,0.058581904,0.012067631,0.0055792257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012700512,0.001809279,0.002356856,0.0022905096,0.0016492632,0.0026733314,0.006137359,0.0024626357,0.004993968],"category_scores_gemma":[0.07616233,0.0015339353,0.0010113458,0.0007666598,0.0029640503,0.0112951975,0.007478928,0.0038897218,0.0007114273],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008105781,0.0031530953,0.053432755,0.0012658412,0.0007085732,0.0029057956,0.0031502312,0.17185333,0.096281804,0.077984266,0.00561338,0.5755452],"study_design_scores_gemma":[0.0007550443,0.0024117825,0.0038538685,0.0001472823,0.00043646464,0.0015750219,0.00049117894,0.80518174,0.07359594,0.107988425,0.0033435638,0.00021964534],"about_ca_topic_score_codex":0.0011981181,"about_ca_topic_score_gemma":0.0021199428,"teacher_disagreement_score":0.012700512,"about_ca_system_score_codex":0.0007889351,"about_ca_system_score_gemma":0.0028258343,"threshold_uncertainty_score":0.06716752},"labels":[],"label_agreement":null},{"id":"W2103954365","doi":"10.1109/icsm.2008.4658102","title":"A requirement-based software testing framework: An industrial practice","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Blackberry (Canada); University of Waterloo","funders":"","keywords":"Computer science; Test strategy; Regression testing; Software engineering; Automation; System integration testing; Software testing; Process (computing); Systems engineering; Software; Software development; Software construction; Engineering","score_opus":0.1849764972952483,"score_gpt":0.3331290527394926,"score_spread":0.14815255544424433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2103954365","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019048078,0.0011136358,0.98474926,0.0041595264,0.00007352907,0.00023944801,0.000035667694,0.0008022914,0.0069218194],"genre_scores_gemma":[0.06961749,0.0013150104,0.9254523,0.0010050829,0.00012361733,0.00050241995,0.00015699262,0.0002559067,0.00157112],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.96882933,0.01702542,0.0020506976,0.0031178342,0.008266873,0.0007097298],"domain_scores_gemma":[0.9565618,0.023992626,0.0020023086,0.009125782,0.007000157,0.0013173404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03971402,0.0022017467,0.0016374369,0.0058706934,0.0019654797,0.007511686,0.0074251005,0.00591395,0.002541858],"category_scores_gemma":[0.03249305,0.0015159963,0.0015484662,0.0043141698,0.01625958,0.010636322,0.003284447,0.008231896,0.0016927498],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000042554046,0.00036204935,0.0011373841,0.0008085091,0.000052289583,0.00042040015,0.0019049661,0.016377699,0.0028824736,0.7713073,0.0063974326,0.19830695],"study_design_scores_gemma":[0.00012509557,0.00046002935,0.001492964,0.0025548802,0.000079524,0.0019762383,0.0010818375,0.15600947,0.008089721,0.641169,0.18676504,0.00019614791],"about_ca_topic_score_codex":0.004530203,"about_ca_topic_score_gemma":0.002028047,"teacher_disagreement_score":0.03971402,"about_ca_system_score_codex":0.004824824,"about_ca_system_score_gemma":0.006259204,"threshold_uncertainty_score":0.21003032},"labels":[],"label_agreement":null},{"id":"W2104487064","doi":"10.1109/wcre.2012.41","title":"Automated Acceptance Testing of JavaScript Web Applications","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"JavaScript; Computer science; Scripting language; Unobtrusive JavaScript; Test script; Web application; World Wide Web; Web testing; Dynamic web page; Web crawler; Web development; Cross-site scripting; Software engineering; Web page; Client-side scripting; Web API; Test case; Programming language; Rich Internet application; Web application security","score_opus":0.04086191583954878,"score_gpt":0.2929715301271169,"score_spread":0.2521096142875681,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104487064","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44485217,0.00014375497,0.5303825,0.000163335,0.00003084193,0.00029452247,0.00013674457,0.02081405,0.0031820047],"genre_scores_gemma":[0.85182786,0.00006286436,0.14527285,0.000058176975,0.000014499637,0.0001677405,0.00027918557,0.00071710633,0.001599629],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9929634,0.0033713419,0.00043374035,0.00063381466,0.002307763,0.00028997735],"domain_scores_gemma":[0.96798986,0.022455761,0.0022078224,0.0037364378,0.0032180476,0.0003920115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022841732,0.0008594504,0.00051618717,0.0007785291,0.0003562791,0.00087516085,0.0012864555,0.0008902976,0.0010325324],"category_scores_gemma":[0.01940321,0.00033649278,0.00049977296,0.00034684318,0.00070026587,0.0008565411,0.0008040913,0.0007011807,0.000505786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011254803,0.0012580341,0.027056322,0.0005880513,0.0001262855,0.0054340074,0.005581543,0.06988224,0.31652734,0.0094498275,0.004394888,0.5585761],"study_design_scores_gemma":[0.00014521107,0.0009780361,0.013594017,0.000118167285,0.00006741591,0.002669445,0.00048027583,0.70131695,0.26415205,0.0068065133,0.009548188,0.00012366382],"about_ca_topic_score_codex":0.0008891759,"about_ca_topic_score_gemma":0.00092019304,"teacher_disagreement_score":0.0022841732,"about_ca_system_score_codex":0.00034583494,"about_ca_system_score_gemma":0.00046298341,"threshold_uncertainty_score":0.012080014},"labels":[],"label_agreement":null},{"id":"W2104555990","doi":"10.1109/sera.2007.127","title":"TODD: Test-Oriented Development and Debugging","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Eclipse; Test-driven development; Debugging; Computer science; Software engineering; Software development; Test (biology); Software development process; Software; Software quality; Programming language","score_opus":0.014663742099982372,"score_gpt":0.253548008777659,"score_spread":0.23888426667767662,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104555990","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006111815,0.0008085612,0.9592866,0.0004103171,0.00029253084,0.00032674638,0.00043185634,0.032252908,0.0055793836],"genre_scores_gemma":[0.026174948,0.0024197875,0.9530417,0.00077779416,0.0002631964,0.0007676399,0.0032042214,0.004980881,0.008369862],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9890884,0.003589567,0.0013301801,0.001158222,0.004248032,0.0005856335],"domain_scores_gemma":[0.9849979,0.0065367725,0.0009791013,0.0039991504,0.0027359785,0.0007510148],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008537647,0.002154233,0.0015046564,0.00446631,0.0010026299,0.0057460796,0.004667973,0.003056959,0.011097203],"category_scores_gemma":[0.029917123,0.0014377978,0.0016282697,0.003548835,0.001901806,0.0049582087,0.0039204336,0.0040746457,0.008157402],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036608664,0.00021730227,0.0018773286,0.002104705,0.00015989067,0.00086133106,0.00056786643,0.008277493,0.014312615,0.07976767,0.08081771,0.8106701],"study_design_scores_gemma":[0.00044401633,0.0003441041,0.0017852006,0.0013061496,0.0002412073,0.0048287464,0.00021834759,0.09954941,0.04969852,0.12068926,0.72059536,0.00029960045],"about_ca_topic_score_codex":0.0015959996,"about_ca_topic_score_gemma":0.0010628332,"teacher_disagreement_score":0.011097203,"about_ca_system_score_codex":0.00088194973,"about_ca_system_score_gemma":0.0023827688,"threshold_uncertainty_score":0.04515195},"labels":[],"label_agreement":null},{"id":"W2104619223","doi":"10.1109/dnsr.2004.1344718","title":"PBit - a pattern-based testing framework for iptables","year":2004,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Firewall (physics); Template; Parameterized complexity; Regression testing; Software; Operating system; Algorithm; Programming language; Software development","score_opus":0.05484419848625403,"score_gpt":0.2950918190262262,"score_spread":0.2402476205399722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104619223","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00089738873,0.00006007573,0.9875021,0.00008614508,0.000020366208,0.00014953682,0.00014691552,0.010015512,0.001122011],"genre_scores_gemma":[0.06534511,0.00024146652,0.9262873,0.0002268524,0.000047074554,0.00074063084,0.0014819116,0.0024561458,0.0031735562],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99405503,0.001589146,0.00075195835,0.00084488414,0.0023659447,0.00039291265],"domain_scores_gemma":[0.993763,0.003001658,0.00055127,0.001437478,0.0010078594,0.00023874236],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053295884,0.0017504934,0.0011613568,0.001986008,0.00073801767,0.0039570834,0.0046376064,0.0019301387,0.007001362],"category_scores_gemma":[0.014025707,0.0011711345,0.0023375088,0.0013530067,0.0020937657,0.0042446577,0.0025567901,0.0028265547,0.0028622265],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058632635,0.000444086,0.0047396133,0.0011016731,0.00027333124,0.0012855565,0.00084792817,0.1139094,0.020086775,0.28959677,0.026140317,0.54098827],"study_design_scores_gemma":[0.00016126115,0.00036289662,0.0008288463,0.00038687955,0.00013886594,0.0015517204,0.0001678308,0.5788017,0.033098776,0.2830365,0.10134489,0.000119945806],"about_ca_topic_score_codex":0.005426115,"about_ca_topic_score_gemma":0.0030190356,"teacher_disagreement_score":0.007001362,"about_ca_system_score_codex":0.0014019728,"about_ca_system_score_gemma":0.0019037529,"threshold_uncertainty_score":0.028185904},"labels":[],"label_agreement":null},{"id":"W2104785339","doi":"10.1109/issre.2003.1251063","title":"Investigating java type analyses for the receiver-classes testing criterion","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; McGill University","keywords":"Java; Computer science; Class (philosophy); Context (archaeology); Set (abstract data type); Intersection (aeronautics); Variable (mathematics); Cover (algebra); Hierarchy; Object (grammar); Software; Code coverage; Type (biology); Programming language; Mathematics; Artificial intelligence","score_opus":0.21090173517988733,"score_gpt":0.39278050659782404,"score_spread":0.1818787714179367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104785339","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39173058,0.00049001427,0.6032285,0.00032726236,0.000023105842,0.000059098005,0.00009704961,0.0019732835,0.0020709892],"genre_scores_gemma":[0.7840868,0.00012408945,0.21482041,0.00008807572,0.00003433103,0.0000394382,0.00016835592,0.0003267464,0.0003117689],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97984076,0.0069541098,0.0011839818,0.0027435282,0.0082898075,0.000987753],"domain_scores_gemma":[0.75131935,0.20527937,0.016110715,0.016274672,0.0104214735,0.00059439935],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015241627,0.00078501855,0.0009360291,0.0022908137,0.0007285897,0.0030427175,0.0022247222,0.0011513879,0.0009992374],"category_scores_gemma":[0.1293447,0.0007778877,0.0013416404,0.001947312,0.002006945,0.004391144,0.0017868214,0.0019316714,0.00026385442],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028093904,0.0004080011,0.1297005,0.00069634843,0.0005683213,0.00045777266,0.002397226,0.2664396,0.05387603,0.05452432,0.0018207183,0.48630187],"study_design_scores_gemma":[0.00018796862,0.0009636162,0.02500905,0.00015805768,0.000279239,0.0010078263,0.00052175525,0.8429037,0.074946344,0.05111484,0.0027662884,0.00014127263],"about_ca_topic_score_codex":0.0029157777,"about_ca_topic_score_gemma":0.0027898212,"teacher_disagreement_score":0.015241627,"about_ca_system_score_codex":0.0016248749,"about_ca_system_score_gemma":0.002245221,"threshold_uncertainty_score":0.08060634},"labels":[],"label_agreement":null},{"id":"W2104838677","doi":"10.1002/spe.1017","title":"Grammar‐based test generation with YouGen","year":2010,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Parsing; Grammar; Natural language processing; Compiler; Rule-based machine translation; Programming language; Artificial intelligence; Generator (circuit theory); Context (archaeology); Linguistics; Power (physics)","score_opus":0.01713707136432013,"score_gpt":0.2785418352168769,"score_spread":0.2614047638525568,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104838677","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019812109,0.0002365904,0.8868251,0.0003430107,0.00012887926,0.0003167787,0.00060233165,0.08440581,0.0073293275],"genre_scores_gemma":[0.28606814,0.00026188986,0.6842024,0.0005747865,0.000055428576,0.00064948463,0.004531194,0.015517992,0.008138744],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99607253,0.0016057957,0.00031772247,0.0005254254,0.0011951752,0.0002833322],"domain_scores_gemma":[0.9897114,0.006586017,0.0004634949,0.0019921998,0.0010689696,0.00017803677],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038172945,0.001073451,0.00062333245,0.001677669,0.00032516077,0.0016645645,0.0018742194,0.0012928803,0.012419],"category_scores_gemma":[0.014654715,0.0008426496,0.0012579586,0.00078971434,0.0015309637,0.002557138,0.0031391212,0.0014989411,0.0037372625],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00089829403,0.0005167669,0.009568476,0.0012888056,0.00027071033,0.0026925162,0.001588453,0.14177714,0.024153072,0.12758212,0.06982379,0.61983985],"study_design_scores_gemma":[0.0004317187,0.00038524586,0.0015941423,0.0003337938,0.00011980978,0.0018856077,0.00030227797,0.67609185,0.080462575,0.118291,0.11992852,0.00017345505],"about_ca_topic_score_codex":0.0008262607,"about_ca_topic_score_gemma":0.000977336,"teacher_disagreement_score":0.012419,"about_ca_system_score_codex":0.0005846895,"about_ca_system_score_gemma":0.0009894656,"threshold_uncertainty_score":0.04154575},"labels":[],"label_agreement":null},{"id":"W2105248826","doi":"10.1109/taicpart.2009.34","title":"Grammar Based Testing of HTML Injection Vulnerabilities in RSS Feeds","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Programming language; XML; Markup language; Rule-based machine translation; Compiler; Code generation; Grammar; Parsing; XHTML; Context (archaeology); Context-free grammar; Syntax; Natural language processing; World Wide Web; Linguistics; Operating system","score_opus":0.028703708936654335,"score_gpt":0.26302400084314165,"score_spread":0.23432029190648732,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105248826","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.66444093,0.00025580538,0.30669296,0.0005161436,0.00006669936,0.0005350556,0.00043159403,0.0223028,0.004757989],"genre_scores_gemma":[0.87134135,0.00011447863,0.12583047,0.00016862917,0.000012859075,0.00014627246,0.0005885422,0.00091689907,0.00088043057],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99244,0.0036825484,0.0005119911,0.0007226739,0.0022049414,0.00043789414],"domain_scores_gemma":[0.97378314,0.019665213,0.0015279041,0.0026742995,0.0020212429,0.00032824895],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003894225,0.00086852204,0.00056379166,0.001978983,0.00044055944,0.0011453246,0.0014693437,0.0017280295,0.0013284317],"category_scores_gemma":[0.024477605,0.00047468595,0.0009484631,0.0010063929,0.0023304338,0.0017476627,0.0018908422,0.00083369645,0.0004934105],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010557637,0.0015675162,0.0647782,0.0010298161,0.00026142356,0.013543004,0.006517923,0.29102218,0.26085454,0.040761255,0.0070029274,0.31160545],"study_design_scores_gemma":[0.00020288932,0.000934173,0.009868461,0.00011179764,0.00017277998,0.0026110867,0.00061998994,0.6730265,0.28649995,0.02001833,0.005827508,0.0001065088],"about_ca_topic_score_codex":0.00186695,"about_ca_topic_score_gemma":0.0015928298,"teacher_disagreement_score":0.003894225,"about_ca_system_score_codex":0.0007574106,"about_ca_system_score_gemma":0.00087217597,"threshold_uncertainty_score":0.020594895},"labels":[],"label_agreement":null},{"id":"W2105821857","doi":"10.1145/566171.566183","title":"Investigating the use of analysis contracts to support fault isolation in object oriented code","year":2002,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Isolation (microbiology); Computer science; Precondition; Design by contract; Software engineering; Reuse; Object-oriented programming; Software; Instrumentation (computer programming); Code (set theory); Object (grammar); Code reuse; Fault detection and isolation; Reliability engineering; Risk analysis (engineering); Programming language; Software system; Engineering; Software construction; Set (abstract data type); Artificial intelligence","score_opus":0.058524941554268906,"score_gpt":0.2645884942805956,"score_spread":0.2060635527263267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105821857","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49609032,0.00047623325,0.49627253,0.0011023355,0.000026553467,0.00033117677,0.000039890565,0.0010220018,0.004638906],"genre_scores_gemma":[0.7931478,0.00029852786,0.20537643,0.000121400146,0.000014200703,0.00012973965,0.000047598365,0.0001282609,0.00073610473],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99052256,0.006125565,0.00034359127,0.0003901262,0.0020814498,0.0005365955],"domain_scores_gemma":[0.9057632,0.075323336,0.007215084,0.006817343,0.004286177,0.00059480494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013097415,0.0005350268,0.00032683843,0.00078416365,0.0007392675,0.0011965169,0.0014666357,0.0012765159,0.00091798016],"category_scores_gemma":[0.053588852,0.00045933176,0.00033057632,0.00096440746,0.0017475403,0.0042760926,0.0013353812,0.0013834968,0.00012997877],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011830094,0.0024420104,0.08303696,0.0011788727,0.00019696714,0.0017592419,0.012642961,0.2162601,0.09561166,0.15482344,0.0017353742,0.42912936],"study_design_scores_gemma":[0.00021440386,0.0020513057,0.014959668,0.00031681277,0.00021708186,0.0009448514,0.0028736698,0.80110925,0.1250678,0.035701346,0.016432723,0.00011116243],"about_ca_topic_score_codex":0.002189618,"about_ca_topic_score_gemma":0.0017646712,"teacher_disagreement_score":0.013097415,"about_ca_system_score_codex":0.00080732047,"about_ca_system_score_gemma":0.0019709067,"threshold_uncertainty_score":0.06926662},"labels":[],"label_agreement":null},{"id":"W2106072155","doi":"10.1145/2568225.2568271","title":"Coverage is not strongly correlated with test suite effectiveness","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":398,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Test suite; Suite; Computer science; Proxy (statistics); Test (biology); Confounding; Test case; Statistics; Data mining; Machine learning; Mathematics; Regression analysis; Geology; Geography","score_opus":0.008925411068723877,"score_gpt":0.22585647833183997,"score_spread":0.2169310672631161,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106072155","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95795864,0.004813873,0.0217529,0.0012511692,0.00009558294,0.00010333019,0.0017359162,0.00071699725,0.011571649],"genre_scores_gemma":[0.99692756,0.0002541098,0.00136371,0.00009800543,0.000048491,0.000030463547,0.0008495739,0.00012218686,0.00030596094],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9697632,0.008215898,0.0030709147,0.0041013393,0.012926863,0.0019217889],"domain_scores_gemma":[0.46632284,0.43638286,0.056389757,0.014442441,0.021483159,0.004978943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012398431,0.0007505367,0.0011979418,0.0048405146,0.0003535021,0.0021181602,0.0011244189,0.0013254515,0.0030667447],"category_scores_gemma":[0.2781792,0.00062934303,0.0011240939,0.003668169,0.0014806554,0.0034885677,0.0015834852,0.0018645943,0.000964816],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005141038,0.00013140621,0.9465784,0.00038419245,0.0011491459,0.0003243153,0.00029994934,0.008851292,0.0023529613,0.000982551,0.001413948,0.037017737],"study_design_scores_gemma":[0.00003195093,0.00055816566,0.97085834,0.00014739187,0.0004636299,0.0012339981,0.00022407377,0.017942064,0.002527745,0.0039787972,0.0019807376,0.00005316856],"about_ca_topic_score_codex":0.002069494,"about_ca_topic_score_gemma":0.0020605656,"teacher_disagreement_score":0.012398431,"about_ca_system_score_codex":0.0009791044,"about_ca_system_score_gemma":0.0008200747,"threshold_uncertainty_score":0.06557},"labels":[],"label_agreement":null},{"id":"W2106459722","doi":"10.48550/arxiv.1212.4123","title":"An Interactive Graph-Based Automation Assistant: A Case Study to Manage the GIPSY's Distributed Multi-tier Run-Time System","year":2012,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Component (thermodynamics); Distributed computing; Graph; Interconnection; Computer network; Theoretical computer science","score_opus":0.0529443423079787,"score_gpt":0.2240369309263044,"score_spread":0.17109258861832569,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106459722","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5769656,0.00030097537,0.35416818,0.0018954837,0.00011251815,0.0008114413,0.00075899326,0.030866846,0.03412001],"genre_scores_gemma":[0.776227,0.00016919833,0.20783848,0.00025190125,0.00003163546,0.0001845448,0.0006214359,0.0018556293,0.012820197],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988971,0.00047357279,0.000042116393,0.00018696874,0.00026612545,0.00013405258],"domain_scores_gemma":[0.99771535,0.0011922738,0.0000979536,0.00042886136,0.00023516007,0.0003303807],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014701372,0.0005866228,0.00033325877,0.00048519394,0.0010974382,0.0015405532,0.0022049178,0.0014579945,0.0048898053],"category_scores_gemma":[0.0030737387,0.00028703164,0.0003982184,0.00065596733,0.001281446,0.0018846733,0.001321142,0.0012093097,0.0012405154],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033786646,0.0028893256,0.024581926,0.0015017813,0.0001956266,0.031316724,0.029504593,0.30729303,0.14624931,0.100309804,0.053383995,0.29939517],"study_design_scores_gemma":[0.00046620643,0.0013695827,0.009768631,0.00010994137,0.00013883135,0.0038818573,0.0050263745,0.6818271,0.08294465,0.020504508,0.19377774,0.00018452287],"about_ca_topic_score_codex":0.0073989225,"about_ca_topic_score_gemma":0.009691643,"teacher_disagreement_score":0.0073989225,"about_ca_system_score_codex":0.0013637727,"about_ca_system_score_gemma":0.000943902,"threshold_uncertainty_score":0.016358078},"labels":[],"label_agreement":null},{"id":"W2106516749","doi":"10.1109/icdcs.1993.287712","title":"Diagnosis of single transition faults in communicating finite state machines","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Set (abstract data type); Fault (geology); State (computer science); Computer science; Finite-state machine; Simple (philosophy); Finite set; Algorithm; Transition system; Transfer (computing); Transition (genetics); Theoretical computer science; Mathematics; Programming language; Parallel computing","score_opus":0.04608729495696682,"score_gpt":0.2611216801873201,"score_spread":0.21503438523035331,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106516749","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055373777,0.0001617832,0.94224167,0.000108904664,0.000032712363,0.000050722418,0.000031841388,0.0014323769,0.0005662579],"genre_scores_gemma":[0.5983258,0.00013392474,0.40031272,0.000066118424,0.00003005026,0.00006532669,0.000133793,0.00006976136,0.0008624723],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989969,0.00029281777,0.00007122933,0.000216404,0.00033059058,0.00009204066],"domain_scores_gemma":[0.99615234,0.002628667,0.0003688434,0.0004622487,0.00032678212,0.00006106266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086470175,0.0006620895,0.00074917794,0.0008237518,0.00042549704,0.00075529085,0.0010868479,0.0010956201,0.00096156],"category_scores_gemma":[0.004789838,0.000253014,0.0006079325,0.00050591375,0.0010577992,0.0013227611,0.0007473051,0.00083268824,0.00017325729],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006557349,0.00014260811,0.0047745216,0.00041526646,0.00011472498,0.0015250271,0.0006963111,0.4760471,0.0415978,0.06648469,0.0023895958,0.40515664],"study_design_scores_gemma":[0.00004061429,0.00009682008,0.00036295637,0.000023679873,0.00002545803,0.000448816,0.000033146018,0.93796694,0.02578387,0.03339443,0.001805708,0.000017629958],"about_ca_topic_score_codex":0.001049797,"about_ca_topic_score_gemma":0.0010118382,"teacher_disagreement_score":0.0010956201,"about_ca_system_score_codex":0.0005886699,"about_ca_system_score_gemma":0.00067823834,"threshold_uncertainty_score":0.004573047},"labels":[],"label_agreement":null},{"id":"W2107915643","doi":"10.1109/tse.2013.20","title":"Whitening SOA Testing via Event Exposure","year":2013,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Service (business); Event (particle physics); Service provider; Information leakage; Implementation; Reliability engineering; Test (biology); Leakage (economics); Embedded system; Real-time computing; Computer security; Software engineering; Engineering","score_opus":0.01368091417226672,"score_gpt":0.2104804074451601,"score_spread":0.1967994932728934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2107915643","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08626238,0.00019500144,0.89985013,0.00024036903,0.000039925726,0.00019239962,0.00006953607,0.009721981,0.003428298],"genre_scores_gemma":[0.77416193,0.00016797896,0.22216786,0.00035535687,0.000044684242,0.00020732328,0.00020884833,0.0008065208,0.0018795562],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942543,0.0021491123,0.0003566314,0.00081496063,0.001995956,0.0004290439],"domain_scores_gemma":[0.9802868,0.011786061,0.0014133053,0.0050396915,0.0011713058,0.00030290938],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031676528,0.0012379392,0.0008654188,0.0018035088,0.00043401754,0.0018685643,0.0015142224,0.0011539634,0.0018118058],"category_scores_gemma":[0.013049924,0.00063532597,0.0012116618,0.0008638807,0.0016651921,0.003044821,0.002651686,0.0014984973,0.00042962152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013682559,0.0008015516,0.011475983,0.00039989257,0.00018795712,0.0018287902,0.001512399,0.1682404,0.16533017,0.039008487,0.002562254,0.60728395],"study_design_scores_gemma":[0.00012635517,0.0005200939,0.002058283,0.00010582913,0.00014099755,0.000802663,0.00013993178,0.6962402,0.24419622,0.04694556,0.008646197,0.0000777066],"about_ca_topic_score_codex":0.00089781696,"about_ca_topic_score_gemma":0.0007814119,"teacher_disagreement_score":0.0031676528,"about_ca_system_score_codex":0.0007307982,"about_ca_system_score_gemma":0.0009861299,"threshold_uncertainty_score":0.016752362},"labels":[],"label_agreement":null},{"id":"W2108367239","doi":"10.1109/hicss.2001.927256","title":"A GUI environment to manipulate FSMs tor testing GUI-based applications in Java","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Java; Graphical user interface; Graphical user interface testing; Programming language; Finite-state machine; Test harness; Representation (politics); Test case; User interface; Software; Software development; User interface design","score_opus":0.03889308573504678,"score_gpt":0.26567195299195556,"score_spread":0.22677886725690877,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2108367239","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062631234,0.00005892939,0.91369367,0.00007872115,0.0000278341,0.00012769664,0.0003156097,0.07646828,0.0029661588],"genre_scores_gemma":[0.18035674,0.00037185324,0.78660065,0.00038642285,0.000068263325,0.0008855617,0.001876319,0.021338584,0.00811566],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99905366,0.00028981498,0.00013018065,0.00013848912,0.0002782038,0.00010965363],"domain_scores_gemma":[0.99631363,0.0026290298,0.00016952615,0.00048469668,0.00024615575,0.00015693415],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019818689,0.0013189876,0.0006802587,0.0012619001,0.00042175758,0.0012092452,0.001889882,0.0011301936,0.012672476],"category_scores_gemma":[0.0053946744,0.0008417022,0.00096007204,0.0004984803,0.0009299177,0.0018919053,0.0016832342,0.0016282339,0.0028078998],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019903409,0.0011132144,0.006445704,0.0015758637,0.00020537383,0.0039329175,0.0032673725,0.05227209,0.20320392,0.07863204,0.050444387,0.59691674],"study_design_scores_gemma":[0.0013220481,0.0012383239,0.0062099337,0.0006642105,0.00019449618,0.003974833,0.0002917969,0.47924158,0.21660295,0.05540042,0.23439325,0.00046616292],"about_ca_topic_score_codex":0.0011861998,"about_ca_topic_score_gemma":0.0012773004,"teacher_disagreement_score":0.012672476,"about_ca_system_score_codex":0.0003805342,"about_ca_system_score_gemma":0.00049331196,"threshold_uncertainty_score":0.042393684},"labels":[],"label_agreement":null},{"id":"W2109277099","doi":"10.1145/2634273","title":"Example-based learning in computer-aided STEM education","year":2014,"lang":"en","type":"article","venue":"Communications of the ACM","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Ryerson University","keywords":"Computer science; Artificial intelligence; Programming language; Software engineering; Human–computer interaction","score_opus":0.060816836911792134,"score_gpt":0.2971825875912495,"score_spread":0.23636575067945736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109277099","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2184547,0.008166802,0.64035374,0.0038529097,0.00024061273,0.00036139786,0.00012834638,0.0018542213,0.12658735],"genre_scores_gemma":[0.6825738,0.002891034,0.3024977,0.00028651798,0.000052627023,0.00010965005,0.00012420476,0.000106746265,0.011357642],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990397,0.00059944374,0.00003184552,0.00008182777,0.00020289337,0.000044222452],"domain_scores_gemma":[0.99743986,0.002044652,0.00006789387,0.00015625564,0.00019379254,0.00009747262],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014837959,0.00019863977,0.00016521299,0.00042790035,0.0003683176,0.0013363353,0.000874536,0.0008302132,0.006502427],"category_scores_gemma":[0.0068495004,0.00017872216,0.00012653139,0.00060145376,0.0007306858,0.002430219,0.00087087177,0.0008846405,0.001024996],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023034838,0.00056958216,0.0019996387,0.0003549928,0.000008563004,0.00012556689,0.0013806564,0.007630503,0.004217854,0.05696702,0.0062284525,0.9202868],"study_design_scores_gemma":[0.0003441122,0.0015559925,0.015811674,0.0014048972,0.0000979428,0.0020217537,0.0042995266,0.22288504,0.05798455,0.47795013,0.21547306,0.00017129064],"about_ca_topic_score_codex":0.0013659761,"about_ca_topic_score_gemma":0.0024378074,"teacher_disagreement_score":0.006502427,"about_ca_system_score_codex":0.00044938165,"about_ca_system_score_gemma":0.0006848981,"threshold_uncertainty_score":0.021752775},"labels":[],"label_agreement":null},{"id":"W2109780335","doi":"10.1109/infcom.1993.253291","title":"Multiple fault diagnosis for finite state machines","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Chinese Academy of Agricultural Sciences","keywords":"Independence (probability theory); Fault (geology); Computer science; Medical diagnosis; Set (abstract data type); State (computer science); Finite-state machine; Algorithm; Reduction (mathematics); Finite set; Simple (philosophy); Theoretical computer science; Mathematics; Programming language","score_opus":0.04364514317896535,"score_gpt":0.2684246354416706,"score_spread":0.22477949226270527,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109780335","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049834345,0.00033388333,0.9931821,0.00009725176,0.000038150385,0.000036544196,0.000019635327,0.00073817815,0.0005707472],"genre_scores_gemma":[0.18967621,0.0003084536,0.80811006,0.00009146925,0.00004913265,0.000113365146,0.00016969486,0.00007060368,0.0014110599],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99855775,0.00041727565,0.00011233638,0.00031200834,0.0004992664,0.00010141239],"domain_scores_gemma":[0.99643964,0.0026773724,0.00022529713,0.00029292348,0.00032042136,0.00004440427],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010715057,0.0008956032,0.0009175613,0.0014724573,0.00049259583,0.0012436132,0.0011521303,0.001285979,0.0022195142],"category_scores_gemma":[0.005218051,0.00037642682,0.0009659873,0.0008327751,0.001212567,0.0018865498,0.0008570313,0.0014315384,0.0004308476],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024288156,0.00008021439,0.0011375302,0.00051092805,0.000121335615,0.000608687,0.00034050338,0.2929566,0.010551,0.20395766,0.0029617406,0.48653102],"study_design_scores_gemma":[0.000038798484,0.00007881463,0.00016933122,0.00007220755,0.000032860713,0.000379176,0.000023534176,0.858048,0.010262018,0.12327169,0.007596266,0.000027212938],"about_ca_topic_score_codex":0.0011587136,"about_ca_topic_score_gemma":0.001122369,"teacher_disagreement_score":0.0022195142,"about_ca_system_score_codex":0.0011766774,"about_ca_system_score_gemma":0.0007416275,"threshold_uncertainty_score":0.008537352},"labels":[],"label_agreement":null},{"id":"W2110029287","doi":"10.1109/ecbs.2005.64","title":"Synthesis of C++ Software from Verifiable CSPm Specifications","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Computer science; Programming language; Formalism (music); Software; Source code; Verifiable secret sharing; Object-oriented programming; Embedded system; Software engineering; Set (abstract data type)","score_opus":0.03881680567377979,"score_gpt":0.24617205225903752,"score_spread":0.20735524658525772,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2110029287","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053382896,0.00005695568,0.9282129,0.00011258098,0.0000392704,0.00022074809,0.00063469214,0.010408942,0.006930994],"genre_scores_gemma":[0.38285094,0.00013511222,0.61010545,0.00008752551,0.000019137418,0.00036847437,0.0013731502,0.0016973937,0.0033628182],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990677,0.00017938646,0.000071977156,0.00011350105,0.0004642085,0.00010319882],"domain_scores_gemma":[0.99735236,0.0013636914,0.00025975725,0.0004425447,0.000522328,0.000059273858],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010739579,0.00065602484,0.0002804908,0.0005191297,0.00041391602,0.00080970925,0.0008603119,0.0005231804,0.003630484],"category_scores_gemma":[0.0048050643,0.00036735518,0.00049076544,0.00049343426,0.0006331804,0.0005075347,0.0005242292,0.0006491347,0.0010628632],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007742539,0.00017059436,0.0023531842,0.0015362038,0.00007090275,0.0024198357,0.0008596048,0.2664489,0.31381702,0.19837533,0.011209499,0.20196472],"study_design_scores_gemma":[0.00019903696,0.00021204556,0.0007192552,0.00011088481,0.00004828747,0.0005216204,0.0000800531,0.5261916,0.41546872,0.024373384,0.03202737,0.00004775341],"about_ca_topic_score_codex":0.0018160258,"about_ca_topic_score_gemma":0.0023292843,"teacher_disagreement_score":0.003630484,"about_ca_system_score_codex":0.0006662729,"about_ca_system_score_gemma":0.0015122634,"threshold_uncertainty_score":0.012145221},"labels":[],"label_agreement":null},{"id":"W2111389084","doi":"10.1109/icc.1996.542187","title":"A test generation tool for specifications in the form of state machines","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Identification (biology); Finite-state machine; Automatic test pattern generation; Heuristic; Minification; Programming language; Test (biology); State (computer science); Test case; Software; Formal specification; Software engineering; Artificial intelligence; Engineering; Machine learning","score_opus":0.10686634083161822,"score_gpt":0.281313890728613,"score_spread":0.1744475498969948,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111389084","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017539466,0.00008082135,0.98375964,0.00005883889,0.00003514607,0.00009626786,0.0002298777,0.012480772,0.0015047191],"genre_scores_gemma":[0.06730985,0.0002181964,0.9226982,0.0002980904,0.00004190913,0.0005850381,0.0019627518,0.002622324,0.0042636627],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9985662,0.0004330786,0.00010808529,0.00020367598,0.0005987158,0.00009020689],"domain_scores_gemma":[0.9975141,0.0016362495,0.00016566584,0.00035338712,0.00027795293,0.000052714742],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013971638,0.0009667036,0.0005796658,0.001398954,0.00041594537,0.00095426926,0.0014013936,0.0010867671,0.009709439],"category_scores_gemma":[0.0064350953,0.00054461905,0.0007461118,0.00095367804,0.0006490577,0.0013507195,0.0010379737,0.001218205,0.0026236598],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032595944,0.00022381553,0.0020549938,0.0012283339,0.00010566551,0.0011692219,0.00041253495,0.03691579,0.0484497,0.06594944,0.05098771,0.7921768],"study_design_scores_gemma":[0.000601679,0.0010085332,0.0021418466,0.0006339712,0.00024845998,0.006972679,0.000148546,0.44318962,0.16899621,0.10868247,0.26718768,0.00018831863],"about_ca_topic_score_codex":0.00067605096,"about_ca_topic_score_gemma":0.0006809105,"teacher_disagreement_score":0.009709439,"about_ca_system_score_codex":0.00032626654,"about_ca_system_score_gemma":0.0008073617,"threshold_uncertainty_score":0.032481313},"labels":[],"label_agreement":null},{"id":"W2111584600","doi":"10.1002/stvr.452","title":"On reducing test length for FSMs with extra states","year":2011,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Tree traversal; Test suite; Computer science; Reduction (mathematics); Set (abstract data type); Algorithm; Finite-state machine; Test set; Implementation; Test case; Theoretical computer science; Mathematics; Programming language; Artificial intelligence","score_opus":0.04516024558319632,"score_gpt":0.25357394988629844,"score_spread":0.20841370430310213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111584600","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26682436,0.00041506076,0.72593534,0.00067201996,0.00006286509,0.00021829447,0.0001364565,0.0032199742,0.002515689],"genre_scores_gemma":[0.64319766,0.00020016685,0.35317352,0.00024218226,0.00009196489,0.00031532452,0.0005145087,0.00051600696,0.0017486374],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99669266,0.0013726215,0.00020033073,0.00034770306,0.0010663036,0.00032042695],"domain_scores_gemma":[0.9642112,0.02630382,0.0025760487,0.0042779506,0.001989783,0.000641047],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021143015,0.00079936493,0.0010178074,0.0010861047,0.0006148038,0.0006593789,0.0016600672,0.0006184493,0.0025915424],"category_scores_gemma":[0.019881576,0.0005111171,0.0010626302,0.00087257125,0.0009857932,0.0025368833,0.0016954124,0.0019228149,0.00037875524],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022940165,0.0005101995,0.0057061682,0.00042544448,0.00013806642,0.0004867791,0.00038125852,0.3548139,0.0684255,0.019762386,0.0030203927,0.5440359],"study_design_scores_gemma":[0.00018369283,0.0010926074,0.0017275421,0.000049481765,0.00007101305,0.00028761823,0.000080660946,0.9344681,0.03316031,0.026220765,0.0026311604,0.000027034246],"about_ca_topic_score_codex":0.0014043475,"about_ca_topic_score_gemma":0.002185384,"teacher_disagreement_score":0.0025915424,"about_ca_system_score_codex":0.0011278823,"about_ca_system_score_gemma":0.0015171719,"threshold_uncertainty_score":0.011181593},"labels":[],"label_agreement":null},{"id":"W2112227385","doi":"10.1109/qsic.2009.30","title":"A Survey of Model-Driven Testing Techniques","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Model-based testing; Non-regression testing; Software reliability testing; Software performance testing; Software engineering; Development testing; Model-driven architecture; Time to market; Software quality; Model-based design; Software development; Software; Reliability engineering; Software construction; Test case; Machine learning; Simulation; Programming language; Engineering","score_opus":0.09180422079795124,"score_gpt":0.3117430841271335,"score_spread":0.21993886332918222,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2112227385","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0067448444,0.06010936,0.91401374,0.0013829197,0.00023914894,0.00034670197,0.0003631088,0.0033543394,0.013445807],"genre_scores_gemma":[0.13670878,0.08552902,0.7649689,0.0011937093,0.00039939667,0.00067539257,0.0016596232,0.0013273042,0.007537967],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9944589,0.0014164964,0.00046121297,0.00052745815,0.0029533722,0.00018259001],"domain_scores_gemma":[0.9873372,0.008979986,0.0006959153,0.0012085995,0.0016546386,0.00012373237],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042213826,0.001968655,0.001534267,0.004262157,0.0005786656,0.0019270794,0.004950191,0.0019054934,0.0036815577],"category_scores_gemma":[0.012154222,0.0012087149,0.002255217,0.0051709535,0.0009813785,0.0028815442,0.0011665879,0.0020697627,0.0015110004],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018646351,0.00028576335,0.00313342,0.0047070854,0.00021704574,0.0006480797,0.00039430644,0.031082597,0.007850688,0.05089102,0.009404941,0.8911986],"study_design_scores_gemma":[0.00031672666,0.00080730824,0.0057193004,0.0068341163,0.0007992837,0.009314699,0.0005836416,0.3413796,0.04528339,0.13718484,0.4514231,0.0003539847],"about_ca_topic_score_codex":0.0027890773,"about_ca_topic_score_gemma":0.0018751518,"teacher_disagreement_score":0.004950191,"about_ca_system_score_codex":0.0012627875,"about_ca_system_score_gemma":0.0018493598,"threshold_uncertainty_score":0.022325039},"labels":[],"label_agreement":null},{"id":"W2113570371","doi":"10.1109/ccece.2005.1557280","title":"A framework for testing distributed software components","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Component-based software engineering; Software construction; System integration testing; White-box testing; Software reliability testing; Software engineering; Common Object Request Broker Architecture; Test harness; Component (thermodynamics); Regression testing; Embedded system; Software development; Operating system; Software","score_opus":0.05397499011663072,"score_gpt":0.2871375926077368,"score_spread":0.2331626024911061,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2113570371","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00071816775,0.00029439415,0.9936103,0.00018629765,0.000034666497,0.00020503571,0.000067299276,0.002038911,0.002844837],"genre_scores_gemma":[0.023247574,0.0004243968,0.97321093,0.00008602679,0.000043381955,0.00057238364,0.00030571801,0.0002455731,0.0018640856],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9941409,0.0018762513,0.0005937483,0.0006470499,0.0024325391,0.0003094066],"domain_scores_gemma":[0.9961747,0.0015072062,0.00025880846,0.0010409552,0.00076512946,0.000253255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005798828,0.0023777005,0.0012787965,0.0027130032,0.001293206,0.0036350365,0.004439279,0.0021494923,0.0040229424],"category_scores_gemma":[0.0072292993,0.0011457348,0.0019334613,0.0017200566,0.0033385837,0.0033887525,0.0024912793,0.0035727338,0.0018944849],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006564627,0.0001811387,0.0006745361,0.00060811767,0.0000714367,0.0006501486,0.000735024,0.056887046,0.010531288,0.7444263,0.007622756,0.17754665],"study_design_scores_gemma":[0.00015560414,0.00040046117,0.00061937066,0.0010338041,0.00012599562,0.0014353154,0.00027248892,0.29070196,0.012938997,0.45654753,0.2355956,0.00017295954],"about_ca_topic_score_codex":0.0061043105,"about_ca_topic_score_gemma":0.004075815,"teacher_disagreement_score":0.0061043105,"about_ca_system_score_codex":0.0020906327,"about_ca_system_score_gemma":0.0040090727,"threshold_uncertainty_score":0.030667543},"labels":[],"label_agreement":null},{"id":"W2114754963","doi":"10.1109/csmr.2002.995799","title":"C to Java migration experiences","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Java; Computer science; Popularity; Real time Java; Source code; Programming language; Java annotation; Generics in Java; strictfp; Operating system; Software engineering","score_opus":0.022324955203872916,"score_gpt":0.27007970476720605,"score_spread":0.24775474956333313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2114754963","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45878693,0.003830282,0.2063803,0.009901011,0.0023122192,0.0009658031,0.001642671,0.031400688,0.28478017],"genre_scores_gemma":[0.6629979,0.003975759,0.17826883,0.0062136403,0.00039207045,0.00047069322,0.0039295293,0.009245362,0.13450618],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983577,0.00048663368,0.000086167274,0.00028378458,0.0005201847,0.00026552196],"domain_scores_gemma":[0.99618715,0.001234312,0.00011733429,0.00083335134,0.0010471424,0.0005807264],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023702076,0.00054277806,0.000297917,0.0007481834,0.001581059,0.002020649,0.0014053468,0.0012288099,0.010576045],"category_scores_gemma":[0.011889389,0.00029362438,0.0003790525,0.0010742218,0.000769843,0.0025328828,0.0024288564,0.0021179616,0.0037146697],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011266125,0.0014326782,0.008020499,0.0007107526,0.000051314102,0.003984327,0.019547991,0.0037925483,0.03396196,0.021265583,0.14944588,0.75665987],"study_design_scores_gemma":[0.00028329776,0.0008496497,0.009683813,0.00038059725,0.00005373455,0.003969352,0.005642157,0.013399425,0.027260056,0.00856982,0.9297605,0.00014757323],"about_ca_topic_score_codex":0.0057247994,"about_ca_topic_score_gemma":0.0083963545,"teacher_disagreement_score":0.010576045,"about_ca_system_score_codex":0.0008449301,"about_ca_system_score_gemma":0.0011107902,"threshold_uncertainty_score":0.035380363},"labels":[],"label_agreement":null},{"id":"W2114770958","doi":"10.1109/qsic.2007.4385506","title":"An Approach to Integration Testing of Object-Oriented Programs","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"Fonds pour la Formation à la Recherche dans l’Industrie et dans l’Agriculture; McMaster University","keywords":"Computer science; Unified Modeling Language; Programming language; Reuse; Object-oriented programming; Test case; Implementation; Software engineering; Object (grammar); Design by contract; Class (philosophy); Component (thermodynamics); Software; Software development; Artificial intelligence; Software construction; Machine learning; Engineering","score_opus":0.04465164295369334,"score_gpt":0.2977340877581621,"score_spread":0.2530824448044688,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2114770958","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029803736,0.0001177761,0.9931791,0.00022716947,0.000027881777,0.00011957789,0.000017111328,0.00055430183,0.0027767557],"genre_scores_gemma":[0.06468149,0.00030371046,0.9315285,0.0002693118,0.00004782419,0.00041316613,0.000109646404,0.00017957938,0.0024668195],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99567956,0.0012563703,0.00029313023,0.00042036365,0.0021714114,0.0001791746],"domain_scores_gemma":[0.9958467,0.0019187273,0.0002772907,0.0009371073,0.0008686959,0.00015142937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002590765,0.0009137479,0.0006586422,0.001908206,0.00082334643,0.0021306297,0.002592525,0.0017152636,0.0022189189],"category_scores_gemma":[0.008477176,0.0006994301,0.0011276846,0.0013988354,0.002494419,0.0023237306,0.0019983733,0.0027829066,0.00051537255],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014091344,0.0005731255,0.0030717906,0.00052577705,0.00013303821,0.0015421925,0.0017087,0.052804008,0.03264087,0.5882175,0.004090478,0.31455165],"study_design_scores_gemma":[0.00019394052,0.00048398532,0.0011920234,0.00042180208,0.000225494,0.002921253,0.0003185416,0.44445434,0.0470112,0.40573323,0.09693474,0.00010954045],"about_ca_topic_score_codex":0.0012173164,"about_ca_topic_score_gemma":0.00096107414,"teacher_disagreement_score":0.002592525,"about_ca_system_score_codex":0.00092227285,"about_ca_system_score_gemma":0.0016400011,"threshold_uncertainty_score":0.013701439},"labels":[],"label_agreement":null},{"id":"W2115479429","doi":"10.1109/issre.2010.39","title":"On the Round Trip Path Testing Strategy","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Tree traversal; Traverse; Computer science; Graph traversal; Tree (set theory); State (computer science); Path (computing); Graph; Algorithm; Theoretical computer science; Mathematics; Combinatorics","score_opus":0.04982608621260986,"score_gpt":0.26784749314140166,"score_spread":0.2180214069287918,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115479429","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014742801,0.00030900378,0.9777337,0.00041990157,0.00001922875,0.00011641208,0.00005785304,0.0006991916,0.005902032],"genre_scores_gemma":[0.35529006,0.0006712059,0.6362531,0.000537051,0.000058304282,0.00032528763,0.00026961608,0.00030825674,0.006287067],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9967726,0.0014465401,0.00015304795,0.0005187794,0.0008757814,0.00023323475],"domain_scores_gemma":[0.9926053,0.00472993,0.0003505396,0.0013347698,0.0007638151,0.00021568198],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023518596,0.0009261178,0.0007231316,0.0016209732,0.0005170763,0.0015115293,0.0019900876,0.0012027205,0.0041166744],"category_scores_gemma":[0.009110017,0.000452788,0.0005893456,0.0014942182,0.0019872172,0.0035074544,0.0015411987,0.0014098899,0.0013587568],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045328296,0.00019207374,0.0023621034,0.00030900174,0.000088334455,0.0006742557,0.0006756177,0.06827269,0.014945058,0.45398024,0.0048296796,0.4532177],"study_design_scores_gemma":[0.00012694705,0.0005941409,0.0011975862,0.0001568882,0.00011186768,0.0015951872,0.0001950912,0.61452967,0.020037962,0.3441466,0.017215313,0.00009278049],"about_ca_topic_score_codex":0.001671109,"about_ca_topic_score_gemma":0.0017537462,"teacher_disagreement_score":0.0041166744,"about_ca_system_score_codex":0.00071137404,"about_ca_system_score_gemma":0.0011046223,"threshold_uncertainty_score":0.013771653},"labels":[],"label_agreement":null},{"id":"W2115962949","doi":"10.1109/hldvt.2009.5340166","title":"Airwolf-TG: A test generator for assertion-based dynamic verification","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; CMC Microsystems (Canada)","funders":"","keywords":"Assertion; Computer science; Automaton; Generator (circuit theory); Test (biology); Programming language; Key (lock); Test case; Software engineering; Theoretical computer science; Machine learning; Operating system","score_opus":0.014793360935907794,"score_gpt":0.27825829845436,"score_spread":0.26346493751845224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115962949","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006582786,0.00011652898,0.92950666,0.00009580944,0.00006470252,0.00022454587,0.0008397459,0.06022379,0.0023453846],"genre_scores_gemma":[0.21900916,0.00018420543,0.76080185,0.00021250782,0.000052887895,0.0008649041,0.0042437734,0.010205903,0.0044247834],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987218,0.00042143816,0.00011021092,0.00024564625,0.00039803403,0.000102913094],"domain_scores_gemma":[0.99547726,0.0028154107,0.00031007882,0.00085519615,0.00043779064,0.00010423957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018856289,0.0011877351,0.00065696926,0.0017005766,0.0003384939,0.0009002464,0.0022793661,0.0012227795,0.013524984],"category_scores_gemma":[0.009596291,0.00065335183,0.00095660693,0.0006138342,0.00094734365,0.0016095709,0.001331857,0.0012040897,0.0037675074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016481,0.00044558768,0.0065420605,0.0016538295,0.00024286045,0.002215856,0.00069936377,0.12473318,0.062393937,0.07515253,0.07580876,0.6484639],"study_design_scores_gemma":[0.00041304625,0.00038233353,0.00070899836,0.00022515305,0.00008443441,0.0015209207,0.00005865906,0.81837124,0.08247752,0.043089744,0.052578487,0.00008949302],"about_ca_topic_score_codex":0.0010905764,"about_ca_topic_score_gemma":0.00090608903,"teacher_disagreement_score":0.013524984,"about_ca_system_score_codex":0.00047990645,"about_ca_system_score_gemma":0.00085119525,"threshold_uncertainty_score":0.045245588},"labels":[],"label_agreement":null},{"id":"W2115973328","doi":"10.1109/icstw.2014.17","title":"Towards a Taxonomy for Simulink Model Mutations","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Taxonomy (biology); Mutation testing; Mutation; clone (Java method); Programming language; Class (philosophy); Artificial intelligence; Operator (biology); Software engineering; Genetics; Biology; Gene","score_opus":0.0638368369646232,"score_gpt":0.2901242805449788,"score_spread":0.2262874435803556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115973328","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023596598,0.001020462,0.9656334,0.0010390271,0.000116297786,0.0005278495,0.00033189575,0.0026013062,0.005133193],"genre_scores_gemma":[0.079365015,0.0010687606,0.91513515,0.00032463943,0.00003958676,0.00039292616,0.00090658397,0.0005577001,0.002209692],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9907816,0.0018611705,0.0020311424,0.0014316203,0.0033080343,0.00058639806],"domain_scores_gemma":[0.98267406,0.0054483977,0.0022035027,0.0037829438,0.0051923813,0.0006987206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062580593,0.0015829848,0.0012046472,0.008596195,0.0030014792,0.0047678156,0.0035054437,0.0032001066,0.0016254925],"category_scores_gemma":[0.016394218,0.0013943757,0.0027903323,0.00439485,0.004539684,0.010102621,0.0029993353,0.0050936043,0.0009657273],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015359372,0.00039162475,0.02224255,0.0010284752,0.000096895754,0.0017456679,0.010023495,0.04491664,0.014580817,0.6753799,0.007854655,0.22158569],"study_design_scores_gemma":[0.00006398199,0.00044507618,0.005608617,0.0015803764,0.00012409693,0.0045262463,0.003832758,0.26567012,0.015057655,0.44170514,0.26110175,0.0002840962],"about_ca_topic_score_codex":0.00759223,"about_ca_topic_score_gemma":0.0068224785,"teacher_disagreement_score":0.008596195,"about_ca_system_score_codex":0.0034921218,"about_ca_system_score_gemma":0.004893282,"threshold_uncertainty_score":0.033096135},"labels":[],"label_agreement":null},{"id":"W2116368564","doi":"10.1145/1985793.1985870","title":"Automated cross-browser compatibility testing","year":2011,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":147,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Compatibility (geochemistry); Computer science; Web browser; Client-side scripting; Emulation; World Wide Web; Backward compatibility; Web-based simulation; Pairwise comparison; Web navigation; Software engineering; Information retrieval; Web page; The Internet; Web API; Operating system; Engineering; Artificial intelligence","score_opus":0.11946833591339981,"score_gpt":0.32342542790760753,"score_spread":0.20395709199420772,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2116368564","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23711525,0.000098132004,0.7491394,0.00017982713,0.000034252134,0.00023978583,0.00019122443,0.010338694,0.002663375],"genre_scores_gemma":[0.81758827,0.00005045582,0.18023378,0.000072345094,0.000010980294,0.00013108489,0.00049260556,0.00048073634,0.0009397947],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9898399,0.0049411226,0.00049818546,0.0012652281,0.0027952935,0.0006602988],"domain_scores_gemma":[0.97505504,0.014755859,0.0017241889,0.00568153,0.0024814068,0.00030199008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00441626,0.0009229603,0.00080024445,0.0014228047,0.0006892781,0.0011196918,0.0019233,0.0012386475,0.0019857225],"category_scores_gemma":[0.0190355,0.00065651454,0.0011191656,0.0006660812,0.0010885083,0.0021403183,0.0029471994,0.0011681842,0.00046636572],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094729586,0.0018896952,0.077136785,0.0006070428,0.00036877766,0.0027330858,0.002921162,0.2550527,0.12204304,0.047095016,0.0050316188,0.48417372],"study_design_scores_gemma":[0.00006305955,0.00027992905,0.0052204677,0.000046526304,0.00006548907,0.0005499489,0.0002416004,0.9196927,0.046003688,0.025331985,0.002458873,0.000045761182],"about_ca_topic_score_codex":0.0023939654,"about_ca_topic_score_gemma":0.0032627005,"teacher_disagreement_score":0.00441626,"about_ca_system_score_codex":0.0006908028,"about_ca_system_score_gemma":0.0017500761,"threshold_uncertainty_score":0.023355663},"labels":[],"label_agreement":null},{"id":"W2117003318","doi":"10.1109/icst.2015.7102595","title":"JSEFT: Automated Javascript Unit Test Generation","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Intel Corporation","keywords":"JavaScript; Unobtrusive JavaScript; Computer science; Oracle; Programming language; Unit testing; Test case; Document Object Model; Web application; Event (particle physics); Rich Internet application; Operating system; Machine learning; Software; XML","score_opus":0.11660022819717285,"score_gpt":0.3053674336741877,"score_spread":0.18876720547701487,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2117003318","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0745409,0.0002302314,0.7947463,0.00019119045,0.00007401634,0.00041295067,0.0017326071,0.1253573,0.0027144903],"genre_scores_gemma":[0.4558674,0.00015352973,0.5322551,0.00022501413,0.000046191723,0.00044235584,0.005969117,0.0033720252,0.00166928],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99790454,0.00057380594,0.00016634745,0.00034426138,0.00085589103,0.00015513496],"domain_scores_gemma":[0.99291396,0.0041691894,0.000602372,0.0010662614,0.0010627946,0.00018541068],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019087865,0.0013539522,0.0007156637,0.0022004722,0.00032084962,0.00089745776,0.0018060027,0.0011696478,0.0033348429],"category_scores_gemma":[0.012372708,0.00045567538,0.0011355646,0.00074147485,0.0007071131,0.0011393641,0.00092191313,0.00069786515,0.001118833],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086679636,0.0006446049,0.02673845,0.0008171862,0.0002216449,0.0012838672,0.00045798483,0.16220045,0.068151124,0.0072714533,0.0349127,0.6964338],"study_design_scores_gemma":[0.00017344805,0.00025141847,0.0042122067,0.00005283821,0.000042133048,0.00075586914,0.0000461798,0.91473097,0.06927861,0.004119308,0.006289582,0.000047359306],"about_ca_topic_score_codex":0.0027599116,"about_ca_topic_score_gemma":0.0022039383,"teacher_disagreement_score":0.0033348429,"about_ca_system_score_codex":0.00060488435,"about_ca_system_score_gemma":0.0013565588,"threshold_uncertainty_score":0.011156201},"labels":[],"label_agreement":null},{"id":"W2118160504","doi":"10.1109/glocom.1990.116807","title":"A novel approach to protocol test sequence generation","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Sequence (biology); Test (biology); Computer science; Finite-state machine; Set (abstract data type); Fault coverage; Test set; Automatic test pattern generation; Theoretical computer science; Fault (geology); Algorithm; Artificial intelligence; Engineering; Programming language","score_opus":0.16318906373876985,"score_gpt":0.312692947955013,"score_spread":0.14950388421624317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118160504","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008585539,0.000043712935,0.99606836,0.00012755088,0.00005133641,0.00017444501,0.00004670419,0.0011277141,0.0015015602],"genre_scores_gemma":[0.045835353,0.00016402804,0.9496802,0.00020309731,0.00007329472,0.00062432676,0.00033533564,0.0002542142,0.0028301664],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9957748,0.001319084,0.00028895724,0.000597008,0.0018052837,0.00021493147],"domain_scores_gemma":[0.9951002,0.00180826,0.00027469773,0.0013128477,0.0013653168,0.00013855893],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022114853,0.0009531217,0.00071158184,0.0015880337,0.000737301,0.0015699756,0.002997974,0.001467275,0.005667935],"category_scores_gemma":[0.0077478634,0.00048176476,0.0009643505,0.0011837245,0.0013035286,0.0025083022,0.0019515918,0.0027666222,0.0017534014],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001929378,0.0003155662,0.0006657342,0.0005126651,0.00009456212,0.0006833361,0.0003140643,0.062088884,0.037584938,0.29941148,0.01443256,0.5837033],"study_design_scores_gemma":[0.00014502583,0.0002964613,0.00020571126,0.00008238245,0.000075563505,0.001046922,0.00009558728,0.7767236,0.043749794,0.12288552,0.054627933,0.00006545971],"about_ca_topic_score_codex":0.0009701882,"about_ca_topic_score_gemma":0.0012144541,"teacher_disagreement_score":0.005667935,"about_ca_system_score_codex":0.00086557306,"about_ca_system_score_gemma":0.0026085107,"threshold_uncertainty_score":0.018961072},"labels":[],"label_agreement":null},{"id":"W2118304037","doi":"10.1109/issre.2002.1173268","title":"A case study using the round-trip strategy for state-based class testing","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Weighting; Random testing; Finite-state machine; Test case; Test strategy; White-box testing; Unified Modeling Language; Model-based testing; Code coverage; Partition (number theory); Fault coverage; Set (abstract data type); Class (philosophy); Context (archaeology); Path (computing); Algorithm; Machine learning; Programming language; Artificial intelligence; Engineering; Mathematics; Software development; Software","score_opus":0.19123208886540766,"score_gpt":0.3524184528424081,"score_spread":0.16118636397700042,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118304037","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.911132,0.00025931283,0.07627536,0.0010542929,0.000059438164,0.0007831365,0.00035174645,0.0005429309,0.009541903],"genre_scores_gemma":[0.92085797,0.00014621623,0.07601564,0.00014949875,0.000016319575,0.00032056362,0.00020421162,0.0000765473,0.0022130336],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99462575,0.0029815498,0.00023477104,0.0004242113,0.0012478047,0.00048591883],"domain_scores_gemma":[0.9783054,0.016116519,0.00080533186,0.0019569676,0.0016202603,0.001195548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041870945,0.0007029437,0.00057060225,0.0012763317,0.0017060149,0.0013304283,0.0022727328,0.0034077389,0.0024107276],"category_scores_gemma":[0.014930848,0.0003512944,0.0009103461,0.0013115262,0.0014398271,0.001686416,0.001064382,0.0015424328,0.0004833109],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004544849,0.01365826,0.11859129,0.0017411972,0.0003992337,0.08172265,0.023368081,0.30034444,0.058943097,0.077317014,0.013875228,0.30549467],"study_design_scores_gemma":[0.0024206277,0.014919755,0.05955274,0.00050142297,0.0005388268,0.02877126,0.016448585,0.6572687,0.11926967,0.033259045,0.06665136,0.00039801555],"about_ca_topic_score_codex":0.008763189,"about_ca_topic_score_gemma":0.011057989,"teacher_disagreement_score":0.008763189,"about_ca_system_score_codex":0.0015094344,"about_ca_system_score_gemma":0.0011325484,"threshold_uncertainty_score":0.022143722},"labels":[],"label_agreement":null},{"id":"W2118505540","doi":"10.1109/tools.2000.868956","title":"Specification-based testing for real-time reactive systems","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Test suite; Metric (unit); Formal specification; Set (abstract data type); Test case; Conformance testing; Suite; System testing; Test strategy; White-box testing; Code coverage; Reliability engineering; Programming language; Software system; Software; Engineering; Machine learning","score_opus":0.07573585341304075,"score_gpt":0.2640879782556397,"score_spread":0.18835212484259894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118505540","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001972221,0.00009684032,0.996949,0.00005695621,0.000007645588,0.000030590203,0.000011804263,0.00057494006,0.00030009676],"genre_scores_gemma":[0.103175096,0.00020876082,0.895322,0.000096085074,0.000022149656,0.00030064108,0.00016139494,0.00021606137,0.0004978545],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98507375,0.005788326,0.0011296978,0.00079078146,0.0068738693,0.00034357468],"domain_scores_gemma":[0.974448,0.01680561,0.002594014,0.0029351574,0.0028557512,0.00036145136],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055985455,0.0010251929,0.000916708,0.0017956131,0.0004270768,0.0019286503,0.0023830747,0.0011420415,0.0016966632],"category_scores_gemma":[0.021248393,0.00058506883,0.0010000736,0.001467583,0.0023677025,0.002473517,0.0011154951,0.0017498037,0.00061100913],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023051143,0.00030883215,0.0039683008,0.0007678227,0.00016189096,0.00038740778,0.0006683826,0.1460948,0.0409056,0.37233424,0.0041786246,0.42999363],"study_design_scores_gemma":[0.00009382002,0.00025579313,0.00074817176,0.00011502011,0.00004893826,0.00045281797,0.000058598747,0.85670424,0.03801665,0.09002846,0.013400136,0.00007735101],"about_ca_topic_score_codex":0.0014319379,"about_ca_topic_score_gemma":0.0018309192,"teacher_disagreement_score":0.0055985455,"about_ca_system_score_codex":0.0011171061,"about_ca_system_score_gemma":0.0013740814,"threshold_uncertainty_score":0.02960831},"labels":[],"label_agreement":null},{"id":"W2118553653","doi":"10.1007/3-540-45785-2_27","title":"Expressing Graphical User’s Input for Test Specifications","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Graphical user interface testing; Graphical user interface; Scalability; Human–computer interaction; Graphical model; Software engineering; Programming language; User experience design; User interface design; Artificial intelligence; Operating system","score_opus":0.05880410136687744,"score_gpt":0.27540840828355595,"score_spread":0.2166043069166785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118553653","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0051992987,0.000073835814,0.93992525,0.00026287095,0.00012607418,0.00011185446,0.0011462186,0.034905244,0.018249426],"genre_scores_gemma":[0.29489836,0.0004250295,0.6302258,0.0008757276,0.00015567531,0.0007170215,0.0066269585,0.026762059,0.03931343],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9978818,0.0007881185,0.00020167223,0.00023259241,0.00069810107,0.00019766274],"domain_scores_gemma":[0.9916005,0.0057039233,0.00028840144,0.0011830336,0.0011002639,0.00012387453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024868797,0.0023326345,0.00075479294,0.0014237346,0.00046569094,0.003409106,0.0019266411,0.002449453,0.0440012],"category_scores_gemma":[0.011818388,0.00097597647,0.0013012537,0.0008586736,0.0012887691,0.0025877908,0.0017694741,0.0018671163,0.013687669],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015482021,0.00029711958,0.002473126,0.002387529,0.00014048166,0.0036506015,0.004914999,0.041536596,0.067537814,0.43032947,0.11221204,0.33297202],"study_design_scores_gemma":[0.00028798284,0.00031592735,0.0008212945,0.0009483602,0.00017809153,0.0021168597,0.00034601075,0.30986068,0.19771527,0.12807389,0.3591637,0.00017195537],"about_ca_topic_score_codex":0.0011230893,"about_ca_topic_score_gemma":0.0011426508,"teacher_disagreement_score":0.0440012,"about_ca_system_score_codex":0.0009592427,"about_ca_system_score_gemma":0.0005651267,"threshold_uncertainty_score":0.14719868},"labels":[],"label_agreement":null},{"id":"W2119042529","doi":"10.1109/iscis.2009.5291883","title":"Using a SAT solver to generate checking sequences","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Model checking; Sequence (biology); Boolean satisfiability problem; Computer science; Solver; Finite-state machine; Satisfiability modulo theories; Set (abstract data type); Algorithm; Theoretical computer science; Abstraction model checking; Automaton; Programming language","score_opus":0.09517591476434714,"score_gpt":0.3385833682943744,"score_spread":0.24340745353002727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119042529","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01324522,0.000058457175,0.9792706,0.0003153719,0.00004486301,0.00025845232,0.00034169303,0.0022494649,0.0042158323],"genre_scores_gemma":[0.064801425,0.000079835925,0.9322905,0.00012869127,0.000016043947,0.00026046182,0.0008042522,0.00024333016,0.0013754234],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99886537,0.00043236944,0.00008660788,0.00022151305,0.0003225354,0.000071618815],"domain_scores_gemma":[0.99421024,0.004828153,0.00022067234,0.0002870339,0.00039885004,0.000055098626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013384217,0.0011777338,0.0005412924,0.0012542378,0.0005623689,0.0008365386,0.0011126656,0.00094775634,0.008548666],"category_scores_gemma":[0.006798356,0.0005952476,0.0010319366,0.001095235,0.00092452567,0.0013561047,0.0007462918,0.001347307,0.0011919024],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003411058,0.0004400644,0.002089387,0.00089793676,0.00013706044,0.0008582886,0.00042888022,0.42150912,0.021981591,0.14839974,0.011865496,0.3910513],"study_design_scores_gemma":[0.00019206852,0.00017163498,0.00018533794,0.00007147425,0.000054216292,0.00023047475,0.000090797286,0.9007817,0.020524317,0.06731776,0.010357431,0.000022832432],"about_ca_topic_score_codex":0.0024550075,"about_ca_topic_score_gemma":0.0047266483,"teacher_disagreement_score":0.008548666,"about_ca_system_score_codex":0.0007409836,"about_ca_system_score_gemma":0.0017674809,"threshold_uncertainty_score":0.02859819},"labels":[],"label_agreement":null},{"id":"W2119582290","doi":"10.1002/spe.520","title":"Investigating the use of analysis contracts to improve the testability of object‐oriented code","year":2003,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":75,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Testability; Precondition; Software engineering; Reuse; Isolation (microbiology); Design by contract; Programming language; Coding (social sciences); Software; Reliability engineering; Software system; Software construction; Engineering","score_opus":0.04497579318473231,"score_gpt":0.31088651099809034,"score_spread":0.265910717813358,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119582290","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.77237654,0.00019960021,0.22135358,0.0008130477,0.000012822801,0.00020736935,0.000031846554,0.001509692,0.0034955423],"genre_scores_gemma":[0.883152,0.00009851744,0.115971826,0.000076593424,0.0000071045947,0.00009116364,0.0000526977,0.00012736562,0.00042266288],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9829012,0.0119570745,0.0006655231,0.00053553685,0.0033021327,0.00063848856],"domain_scores_gemma":[0.8122732,0.14835998,0.015177111,0.015331079,0.0079312995,0.0009273283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022436243,0.00050748757,0.00033672526,0.0011034876,0.00069243443,0.0014131247,0.0014016905,0.0011243257,0.0010335861],"category_scores_gemma":[0.094804965,0.00048533693,0.00030419714,0.0009876809,0.0020905542,0.0037869231,0.0019140118,0.0011003935,0.00015598771],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014045106,0.0023404728,0.122245684,0.00055376586,0.0001302594,0.0010724317,0.01099294,0.1443644,0.069315635,0.07199991,0.0021720007,0.573408],"study_design_scores_gemma":[0.00025976406,0.0023529602,0.035006717,0.00026142417,0.00016970655,0.0008781907,0.0020358774,0.81400406,0.104332164,0.028070979,0.012521769,0.000106317486],"about_ca_topic_score_codex":0.0023475338,"about_ca_topic_score_gemma":0.0013877889,"teacher_disagreement_score":0.022436243,"about_ca_system_score_codex":0.0012765683,"about_ca_system_score_gemma":0.0020214482,"threshold_uncertainty_score":0.11865562},"labels":[],"label_agreement":null},{"id":"W2120359737","doi":"10.1109/scam.2008.29","title":"Automated Migration of List Based JSP Web Pages to AJAX","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Ajax; Web page; Computer science; World Wide Web; Static web page; Dynamic web page; Page view; Web application; Web development","score_opus":0.03209367110724004,"score_gpt":0.26569929216042576,"score_spread":0.23360562105318572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120359737","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38650408,0.0002634458,0.37650707,0.0005895174,0.0005167474,0.0007733575,0.000707401,0.21743488,0.016703425],"genre_scores_gemma":[0.6699127,0.00021404515,0.290503,0.00038944941,0.000068058675,0.00032653462,0.0017149035,0.013341719,0.023529539],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988636,0.00021168537,0.00008958742,0.0001815098,0.00049106457,0.00016259559],"domain_scores_gemma":[0.9952272,0.0014000561,0.00027498262,0.00195458,0.0008934393,0.00024964567],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011932577,0.0007180219,0.00048533228,0.00084553135,0.0011207798,0.0021930335,0.0016841671,0.0006956115,0.0040434124],"category_scores_gemma":[0.0066278665,0.0006724964,0.00049040443,0.0007147633,0.00079418905,0.0018161729,0.0016344415,0.00145249,0.0020200412],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038201264,0.0013274173,0.015545181,0.00047274583,0.00014188807,0.00530995,0.0064270385,0.019127656,0.30547845,0.017701393,0.058701795,0.56594634],"study_design_scores_gemma":[0.00036202208,0.0008477286,0.017162075,0.000117655785,0.00012255613,0.0018678285,0.0013227002,0.29348418,0.53473765,0.0137713095,0.13599503,0.00020924384],"about_ca_topic_score_codex":0.0028830927,"about_ca_topic_score_gemma":0.0025918644,"teacher_disagreement_score":0.0040434124,"about_ca_system_score_codex":0.0005971666,"about_ca_system_score_gemma":0.00080610206,"threshold_uncertainty_score":0.013526499},"labels":[],"label_agreement":null},{"id":"W2121316343","doi":"10.1109/iccd.2006.4380831","title":"Adding Debug Enhancements to Assertion Checkers for Hardware Emulation and Silicon Debug","year":2006,"lang":"en","type":"article","venue":"Proceedings, IEEE International Conference on Computer Design/Proceedings - IEEE International Conference on Computer Design","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Debugging; Emulation; Assertion; Computer science; Background debug mode interface; Hardware emulation; Embedded system; Computer architecture; Software bug; Programming language; Software; Field-programmable gate array; Psychology","score_opus":0.12912903322472305,"score_gpt":0.33324142009886076,"score_spread":0.2041123868741377,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121316343","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010613626,0.00010803274,0.9780548,0.00013517209,0.000091318034,0.000072771145,0.000057874146,0.009804926,0.0010614936],"genre_scores_gemma":[0.23237514,0.00017228656,0.76266235,0.00022034839,0.0000932715,0.00011294918,0.0002670342,0.0015726424,0.0025238965],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9965313,0.001215649,0.00034287642,0.00042055413,0.001246739,0.00024290287],"domain_scores_gemma":[0.97885865,0.011156235,0.0018548642,0.005766503,0.0021706747,0.00019313037],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003655389,0.0012113082,0.00052585214,0.0017757034,0.00037366702,0.0012238707,0.0018927339,0.000849115,0.004368544],"category_scores_gemma":[0.02095746,0.0007500827,0.0008768101,0.0006805848,0.00077653676,0.004050256,0.0015425443,0.0024894965,0.0010338491],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009595282,0.00043817537,0.01025046,0.0007080605,0.00015388739,0.0010897706,0.000748218,0.0461598,0.13718042,0.10297678,0.008241559,0.6910934],"study_design_scores_gemma":[0.00021993076,0.00088923005,0.002829608,0.00028876177,0.00026937848,0.0019651244,0.000116737414,0.49267095,0.39185223,0.041562125,0.06717317,0.00016276437],"about_ca_topic_score_codex":0.00068057736,"about_ca_topic_score_gemma":0.0016349163,"teacher_disagreement_score":0.004368544,"about_ca_system_score_codex":0.00048177663,"about_ca_system_score_gemma":0.0009472783,"threshold_uncertainty_score":0.019331753},"labels":[],"label_agreement":null},{"id":"W2121450025","doi":"10.1109/sera.2008.29","title":"Modeling Enhanced Scenarios for Automated Instrumentation","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Traceability; Computer science; Scalability; Software engineering; Instrumentation (computer programming); Model-based testing; Scenario testing; Automation; Focus (optics); Semantics (computer science); Test case; Dominance (genetics); Systems engineering; Programming language; Engineering; Artificial intelligence; Machine learning; Database","score_opus":0.04232699505102225,"score_gpt":0.29299683743677946,"score_spread":0.2506698423857572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121450025","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0082824705,0.00015234279,0.982152,0.0003518036,0.000023785804,0.00025411384,0.00017628798,0.0007089258,0.00789825],"genre_scores_gemma":[0.22444063,0.0005440949,0.7694814,0.00015668043,0.0000324639,0.00084900274,0.0006260049,0.0002837035,0.0035860902],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996329,0.0020426444,0.00027981534,0.0003339323,0.000776036,0.00023853686],"domain_scores_gemma":[0.99407464,0.0033804511,0.00052552595,0.0013709341,0.00043390822,0.00021447409],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003588044,0.0010227734,0.00047189713,0.0016170718,0.0006394939,0.0026332592,0.0019069115,0.0020591188,0.0069710575],"category_scores_gemma":[0.010457114,0.00080825685,0.0019606443,0.0012241732,0.002445994,0.0056850794,0.0037261986,0.0022326557,0.0010333867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049216364,0.000066917266,0.0008858037,0.00010772729,0.000024772171,0.00041993213,0.00058217574,0.19673008,0.0020474147,0.7767769,0.0009931214,0.021315856],"study_design_scores_gemma":[0.000038114376,0.000064396416,0.00022281513,0.00015103897,0.000020241485,0.00030162212,0.00013896103,0.5649705,0.0023436442,0.39915028,0.032570776,0.000027616192],"about_ca_topic_score_codex":0.0028781442,"about_ca_topic_score_gemma":0.002557576,"teacher_disagreement_score":0.0069710575,"about_ca_system_score_codex":0.0011570585,"about_ca_system_score_gemma":0.0017070319,"threshold_uncertainty_score":0.023320496},"labels":[],"label_agreement":null},{"id":"W2121504314","doi":"10.1145/2642937.2642991","title":"Leveraging existing tests in automated test generation for web applications","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Test suite; Computer science; Web crawler; Crawling; Code coverage; Test (biology); Web testing; Test harness; Random testing; Test Management Approach; Test case; Domain (mathematical analysis); Generator (circuit theory); Software engineering; Event (particle physics); Data mining; Machine learning; Web page; Programming language; World Wide Web; Software; Power (physics); Web development; Software development; Web application security","score_opus":0.060793006232949295,"score_gpt":0.31997880997381783,"score_spread":0.2591858037408685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121504314","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09726414,0.00082251057,0.86962104,0.00051748694,0.00005613618,0.0004568061,0.0003810375,0.0279218,0.0029590183],"genre_scores_gemma":[0.5099316,0.00042649976,0.48478815,0.00033558195,0.00004988772,0.00035147613,0.0016652038,0.0015063904,0.00094521366],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9882414,0.0060722283,0.00081405905,0.0013201314,0.0030893586,0.0004628009],"domain_scores_gemma":[0.91893536,0.058327645,0.0049412875,0.0112598995,0.00589492,0.0006409101],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0071261465,0.001631827,0.0010531268,0.005579936,0.00060581,0.0017529356,0.002421212,0.001447701,0.0019262121],"category_scores_gemma":[0.053519756,0.00087318796,0.0011193793,0.0023551378,0.0015296865,0.0026982762,0.001530791,0.0013561603,0.0008156699],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045947396,0.0009827035,0.021774232,0.00081775757,0.0002508628,0.0013035404,0.00085926655,0.13656256,0.03660092,0.0063666133,0.0052762236,0.78874594],"study_design_scores_gemma":[0.0001819519,0.00062950706,0.0072091175,0.00034045515,0.00018964117,0.0011318146,0.00022198084,0.893377,0.06869554,0.01838878,0.009519825,0.000114374525],"about_ca_topic_score_codex":0.0028065343,"about_ca_topic_score_gemma":0.004962389,"teacher_disagreement_score":0.0071261465,"about_ca_system_score_codex":0.0009592399,"about_ca_system_score_gemma":0.0018589062,"threshold_uncertainty_score":0.037687123},"labels":[],"label_agreement":null},{"id":"W2121513851","doi":"10.1007/978-3-319-22689-7_35","title":"Hermes: A Targeted Fuzz Testing Framework","year":2015,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Software security assurance; Fuzz testing; Computer science; Security testing; Test (biology); Software; Test case; Code (set theory); Variety (cybernetics); Buffer overflow; Computer security; Information security; Security information and event management; Security service; Cloud computing security; Artificial intelligence; Programming language; Set (abstract data type); Machine learning; Cloud computing; Operating system","score_opus":0.10195110765609292,"score_gpt":0.32900234293158076,"score_spread":0.22705123527548784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121513851","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023194137,0.00023253946,0.9781819,0.00018379552,0.0000717442,0.00007430679,0.0001404181,0.012414607,0.006381308],"genre_scores_gemma":[0.16359538,0.0007493052,0.8071736,0.0006876361,0.00014144008,0.00026916587,0.0007780658,0.0050218366,0.021583542],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99795413,0.000417616,0.00009488784,0.00030820837,0.0010625125,0.00016256084],"domain_scores_gemma":[0.9983553,0.000764115,0.000087018394,0.00047989463,0.000255924,0.000057697765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022170858,0.0013630845,0.0007980517,0.0014277359,0.0005207423,0.002142088,0.003123369,0.0014673462,0.009531349],"category_scores_gemma":[0.004720049,0.00080681714,0.0013478544,0.0006557107,0.0017215705,0.003608292,0.00251454,0.0025590786,0.0027361833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025137624,0.00016766058,0.0009160258,0.0004438047,0.00010397295,0.0003627943,0.00037994492,0.050139055,0.01571298,0.46935672,0.028409135,0.43375656],"study_design_scores_gemma":[0.0000723101,0.00011475687,0.00040937128,0.00021986142,0.000106840176,0.00064247596,0.000060021535,0.3433981,0.027962087,0.537091,0.0898482,0.00007492582],"about_ca_topic_score_codex":0.001556791,"about_ca_topic_score_gemma":0.0019936431,"teacher_disagreement_score":0.009531349,"about_ca_system_score_codex":0.0007265671,"about_ca_system_score_gemma":0.0012499546,"threshold_uncertainty_score":0.031885564},"labels":[],"label_agreement":null},{"id":"W2122097722","doi":"10.1109/ftcs.1996.534610","title":"A framework for conformance testing of systems communicating through rendezvous","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Conformance testing; Computer science; Rendezvous; Set (abstract data type); Fault (geology); Test case; Fault coverage; Test (biology); System under test; Finite-state machine; Model-based testing; Reliability engineering; System testing; Non-regression testing; Test suite; Code coverage; Algorithm; Programming language; Engineering; Software; Machine learning; Software system; Standardization","score_opus":0.15028303002887528,"score_gpt":0.32538206358263977,"score_spread":0.1750990335537645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122097722","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00089894317,0.0001356767,0.99773633,0.00007700002,0.00001616072,0.00010262494,0.00003982031,0.00032987446,0.00066363584],"genre_scores_gemma":[0.088213,0.0006033102,0.9075683,0.00023044736,0.00020319286,0.0013153506,0.00045654672,0.00027157326,0.0011383683],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98655266,0.0049408223,0.0012935003,0.0018542306,0.0045217765,0.0008370195],"domain_scores_gemma":[0.98354244,0.010899542,0.0011564698,0.002392603,0.001676036,0.00033282896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008804382,0.0027334199,0.0021818266,0.0046240217,0.0013159134,0.0032009978,0.0051710517,0.0026780765,0.0033803196],"category_scores_gemma":[0.0202399,0.0012439631,0.0040523033,0.0029025243,0.0072587426,0.004538565,0.0032368447,0.0053760065,0.0011125627],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006441326,0.00013676297,0.00055725186,0.0003203463,0.000075773125,0.0005455257,0.00038085337,0.124690264,0.006046587,0.8184268,0.0015549446,0.047200497],"study_design_scores_gemma":[0.00008408255,0.00031194164,0.00038920893,0.00022674637,0.000094226925,0.00079439365,0.00008780478,0.3819051,0.005458973,0.5930618,0.017495176,0.00009058786],"about_ca_topic_score_codex":0.0036411297,"about_ca_topic_score_gemma":0.0018358887,"teacher_disagreement_score":0.008804382,"about_ca_system_score_codex":0.0025988393,"about_ca_system_score_gemma":0.0023877532,"threshold_uncertainty_score":0.046562552},"labels":[],"label_agreement":null},{"id":"W2122167693","doi":"10.4028/www.scientific.net/amm.241-244.2718","title":"The Quantified Evaluation Method of Project Test Based on Multi-Computing","year":2012,"lang":"en","type":"article","venue":"Applied Mechanics and Materials","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Tellabs (Canada)","funders":"","keywords":"Test (biology); Computer science; Project management; Grid; Project management triangle; Quality (philosophy); Systems engineering; Project planning; Test case; Software; Engineering management; Engineering","score_opus":0.06886701825330585,"score_gpt":0.34737752608012773,"score_spread":0.27851050782682185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122167693","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034562636,0.00030636322,0.9560988,0.00014720019,0.000063813255,0.0004399081,0.00011514068,0.0005292457,0.007736844],"genre_scores_gemma":[0.54517883,0.00032255013,0.45094058,0.00006509242,0.000039377475,0.0010107054,0.00023312336,0.00013297176,0.002076749],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98174083,0.006966598,0.0011795689,0.0015319077,0.008120905,0.00046026908],"domain_scores_gemma":[0.9855278,0.0055561713,0.0016619292,0.001181476,0.005611315,0.0004613328],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005906769,0.001354746,0.00082073454,0.007154228,0.0010242298,0.0026442334,0.0013381247,0.0006623839,0.0020479623],"category_scores_gemma":[0.021689612,0.00031996073,0.0008164148,0.0041743517,0.0016897998,0.003117694,0.002065773,0.0008944831,0.00033662532],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004212274,0.00034063822,0.02894941,0.0010959321,0.00022580633,0.0002192275,0.0024005838,0.039608292,0.02602991,0.078444265,0.0029212565,0.81934345],"study_design_scores_gemma":[0.0001462227,0.0020379252,0.05796331,0.0007011437,0.0004423665,0.0016784888,0.0032739434,0.743404,0.06701288,0.09032116,0.03249209,0.0005264888],"about_ca_topic_score_codex":0.0028125169,"about_ca_topic_score_gemma":0.002071547,"teacher_disagreement_score":0.007154228,"about_ca_system_score_codex":0.001789959,"about_ca_system_score_gemma":0.0023783399,"threshold_uncertainty_score":0.031238377},"labels":[],"label_agreement":null},{"id":"W2122170536","doi":"10.1109/cmpsac.1997.625060","title":"Natural optimization algorithms for optimal regression testing","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Mathematical optimization; Algorithm; Simulated annealing; Regression testing; Integer programming; Software; Mathematics; Software system","score_opus":0.07023088735804665,"score_gpt":0.2908348467376765,"score_spread":0.22060395937962984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122170536","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0025470904,0.0004382091,0.99397814,0.00017757025,0.000032285898,0.000044314354,0.000035626774,0.0002323993,0.002514377],"genre_scores_gemma":[0.15251915,0.0008300213,0.84175104,0.0002793626,0.00010932488,0.00070518,0.00031099672,0.00021710231,0.003277861],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979315,0.0010960251,0.00010246291,0.0003318703,0.00040530547,0.00013287003],"domain_scores_gemma":[0.99464923,0.0042479984,0.0003125422,0.00026395556,0.00045106062,0.00007517533],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030107398,0.001305764,0.001336805,0.0016224218,0.0006598982,0.0012659675,0.0014024362,0.0015448205,0.004398937],"category_scores_gemma":[0.011716939,0.0006802687,0.0010745018,0.0014273708,0.0018834288,0.0016516361,0.0013471551,0.0021103565,0.0008027378],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007463696,0.00007611304,0.0003413786,0.00014971467,0.000053982527,0.000037063164,0.000068428315,0.81532186,0.00051436434,0.10684506,0.0027720751,0.07374529],"study_design_scores_gemma":[0.000038377635,0.000027888498,0.00007084794,0.00001880134,0.000008245349,0.000016618524,0.000010563718,0.91745114,0.00019121866,0.08041246,0.0017440646,0.000009689747],"about_ca_topic_score_codex":0.0032030907,"about_ca_topic_score_gemma":0.0028001186,"teacher_disagreement_score":0.004398937,"about_ca_system_score_codex":0.0016890673,"about_ca_system_score_gemma":0.0017900314,"threshold_uncertainty_score":0.015922487},"labels":[],"label_agreement":null},{"id":"W2122175577","doi":"10.1109/iri-05.2005.1506535","title":"A high level design and testing of celestial tracking software with the focus of reusability","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trent University","funders":"","keywords":"Reusability; Computer science; Constraint (computer-aided design); Dependency (UML); Software; Focus (optics); Process (computing); Partition (number theory); Domain (mathematical analysis); Software engineering; Systems engineering; Programming language; Engineering; Mathematics","score_opus":0.06748813021111748,"score_gpt":0.2572179153920251,"score_spread":0.1897297851809076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122175577","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03388951,0.00006876154,0.96176195,0.00009128958,0.000013533019,0.00017918313,0.000027723894,0.0024212284,0.0015468119],"genre_scores_gemma":[0.36677626,0.00010426792,0.62973523,0.000091379225,0.000015183448,0.00031494786,0.00018462684,0.00041135657,0.0023668346],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99677664,0.0010526908,0.00019714728,0.0004901582,0.0012585355,0.00022473503],"domain_scores_gemma":[0.9963278,0.0015690549,0.00029208252,0.0009061451,0.00078144873,0.00012355232],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002291774,0.00071327476,0.00058129523,0.00086337666,0.00044740565,0.0014877135,0.0017726085,0.001139305,0.0022864183],"category_scores_gemma":[0.0067051332,0.000553907,0.0009187015,0.00036965325,0.0012411689,0.0013796365,0.0006026441,0.000983395,0.000554773],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060344016,0.0005655193,0.008884637,0.0010329862,0.00020200497,0.0023770595,0.0016677931,0.12538564,0.38260707,0.07875841,0.0021883259,0.3957271],"study_design_scores_gemma":[0.00019481739,0.0016623386,0.0036699362,0.0002478516,0.00021576375,0.0019013783,0.00018850324,0.68531835,0.26231778,0.024432663,0.019757118,0.000093569935],"about_ca_topic_score_codex":0.0014698755,"about_ca_topic_score_gemma":0.0009510017,"teacher_disagreement_score":0.002291774,"about_ca_system_score_codex":0.00064378715,"about_ca_system_score_gemma":0.0011212608,"threshold_uncertainty_score":0.012120187},"labels":[],"label_agreement":null},{"id":"W2122260303","doi":"10.1109/rtcsa.1999.811206","title":"Fault coverage in testing real-time systems","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Université de Montréal","funders":"","keywords":"Computer science; Fault coverage; Test suite; Reliability engineering; Fault (geology); Real-time computing; Software quality; Suite; Software fault tolerance; Software; Fault model; System under test; System testing; Test case; Software development; Engineering; Software engineering; Programming language","score_opus":0.025980541671107846,"score_gpt":0.2526553932488752,"score_spread":0.22667485157776734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122260303","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20760615,0.0015051573,0.7863705,0.00048346876,0.000040629828,0.000053809403,0.0002525858,0.00076302723,0.0029247631],"genre_scores_gemma":[0.9548287,0.0004906183,0.043476794,0.000064895794,0.00005889589,0.00010687352,0.000318251,0.00011485686,0.00054017856],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9927528,0.0028967625,0.0005081051,0.0008744363,0.002455202,0.00051274704],"domain_scores_gemma":[0.96139985,0.031996086,0.00210775,0.0023365088,0.0017426597,0.00041712937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039804624,0.0009420252,0.0010287312,0.003991861,0.00048900757,0.0015740523,0.0015698096,0.0015384832,0.0015145217],"category_scores_gemma":[0.029562475,0.0005191653,0.0011073785,0.0022873702,0.0025231866,0.0031835139,0.0011911613,0.00093849417,0.00019553369],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039297907,0.00017418081,0.018258022,0.0004457948,0.00018010709,0.0007090698,0.0006443194,0.84176475,0.011077442,0.066902556,0.0007915433,0.05865924],"study_design_scores_gemma":[0.000030509049,0.00020197421,0.0018337099,0.00007462157,0.000057615478,0.00030627538,0.00008462266,0.9419209,0.0057178535,0.049065426,0.00068630796,0.000020191748],"about_ca_topic_score_codex":0.0069719255,"about_ca_topic_score_gemma":0.002105505,"teacher_disagreement_score":0.0069719255,"about_ca_system_score_codex":0.0018823704,"about_ca_system_score_gemma":0.00096065534,"threshold_uncertainty_score":0.02105093},"labels":[],"label_agreement":null},{"id":"W2122282651","doi":"10.1109/scam.2002.1134113","title":"Predicate-based dynamic slicing of message passing programs","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Program slicing; Predicate (mathematical logic); Computer science; Slicing; Predicate variable; Computation; Programming language; Algorithm; Theoretical computer science","score_opus":0.016900788852369494,"score_gpt":0.25476644214893224,"score_spread":0.23786565329656273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122282651","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047984887,0.00007433148,0.9487724,0.000050620776,0.0000084044,0.00006124203,0.00007961427,0.002142216,0.0008262193],"genre_scores_gemma":[0.5014664,0.00012584782,0.49675295,0.00003377317,0.00001899413,0.0000893259,0.00041297651,0.00035339623,0.000746349],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99932027,0.00014723654,0.000049244743,0.00010966637,0.00027251727,0.00010108741],"domain_scores_gemma":[0.998007,0.00091436773,0.00026431956,0.00038278988,0.00035187678,0.000079622696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010277638,0.00051126326,0.0005002274,0.00072477193,0.00043846818,0.00068149745,0.0006164417,0.00027798035,0.0010986757],"category_scores_gemma":[0.003266406,0.0002751676,0.00051664346,0.00060848723,0.0010308989,0.0015744099,0.0007490227,0.00051650585,0.00017746216],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009946536,0.00011521311,0.0069614216,0.0003204933,0.00006466087,0.0004411064,0.0009236128,0.35953027,0.11529857,0.10411121,0.0026373924,0.40860137],"study_design_scores_gemma":[0.000031015526,0.00010769978,0.0008142731,0.000021052623,0.000029862402,0.00011887561,0.000053218588,0.88380307,0.083735354,0.027778404,0.0034882003,0.000019024721],"about_ca_topic_score_codex":0.0031109848,"about_ca_topic_score_gemma":0.0027162419,"teacher_disagreement_score":0.0031109848,"about_ca_system_score_codex":0.00078849687,"about_ca_system_score_gemma":0.0010791699,"threshold_uncertainty_score":0.0061857104},"labels":[],"label_agreement":null},{"id":"W2122626119","doi":"10.1145/1295074.1295086","title":"Regression test suite reduction using extended dependence analysis","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Extended finite-state machine; Test suite; Computer science; Reduction (mathematics); Regression analysis; Set (abstract data type); Finite-state machine; Suite; Test case; Algorithm; Machine learning; Mathematics; Programming language","score_opus":0.03315978076932156,"score_gpt":0.32409144525396816,"score_spread":0.2909316644846466,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122626119","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060600206,0.00012539401,0.9328026,0.0001323801,0.000021894448,0.00013103338,0.00016502207,0.004727084,0.001294306],"genre_scores_gemma":[0.5756406,0.000113043694,0.42041206,0.00010997334,0.000030356001,0.0003501548,0.0011571479,0.00054888753,0.0016378209],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9957938,0.0016400155,0.00021262791,0.0004646236,0.0016451677,0.00024374912],"domain_scores_gemma":[0.99115026,0.00501876,0.0007182826,0.0015610991,0.001412182,0.00013943267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012954837,0.0010500702,0.0011399704,0.0024344434,0.00031414762,0.00055071636,0.0014327235,0.00050300587,0.0018212594],"category_scores_gemma":[0.012028547,0.0004717615,0.0017827419,0.0010938498,0.00054127286,0.0010110755,0.00126372,0.0013849022,0.00033360778],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004989033,0.0006396967,0.0077500404,0.0002526976,0.00024691876,0.0007930695,0.00018500364,0.44186574,0.056855194,0.013810529,0.0034966415,0.4736055],"study_design_scores_gemma":[0.000032056738,0.00013834644,0.0010795589,0.00001322681,0.000046305133,0.00018571754,0.000012964401,0.978243,0.011944586,0.007364543,0.00092028215,0.000019417119],"about_ca_topic_score_codex":0.002244935,"about_ca_topic_score_gemma":0.002112427,"teacher_disagreement_score":0.0024344434,"about_ca_system_score_codex":0.0006479989,"about_ca_system_score_gemma":0.0011669635,"threshold_uncertainty_score":0.0068511963},"labels":[],"label_agreement":null},{"id":"W2122715775","doi":"10.1007/s12243-014-0449-0","title":"Integration testing of communicating systems with unknown components","year":2014,"lang":"en","type":"article","venue":"Annals of Telecommunications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Modular design; Reachability; Component (thermodynamics); Model-based testing; TRACE (psycholinguistics); Isolation (microbiology); Model checking; Asynchronous communication; System under test; Inference; Test case; Theoretical computer science; Algorithm; Artificial intelligence; Machine learning; Programming language","score_opus":0.14085710865973636,"score_gpt":0.33044936925932655,"score_spread":0.1895922605995902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122715775","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3874103,0.0004564639,0.608238,0.00030097226,0.00004593883,0.000048474776,0.000030289442,0.0014946997,0.0019748297],"genre_scores_gemma":[0.93908334,0.00011853521,0.060019914,0.000041458858,0.000025001684,0.000027972694,0.000050283812,0.000097579934,0.00053588697],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9915671,0.003801485,0.00037762025,0.00064858596,0.0030000252,0.0006051817],"domain_scores_gemma":[0.95440876,0.03797199,0.0019687505,0.0032991564,0.0019399216,0.0004113207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037549536,0.0010700694,0.00087991427,0.0010953868,0.0005034301,0.0012746892,0.0021974235,0.0014034704,0.0010960778],"category_scores_gemma":[0.030297391,0.00065296213,0.00083495857,0.00091467856,0.0020317691,0.0024520608,0.0015796397,0.0015726365,0.00018167777],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002412137,0.0010101582,0.033893917,0.0007646951,0.000550707,0.0034106665,0.0021396137,0.45468643,0.08854794,0.06866072,0.0015847705,0.34233823],"study_design_scores_gemma":[0.00010143085,0.00039600144,0.002441299,0.0000693694,0.00015845157,0.00054690015,0.00009899196,0.93023986,0.030957375,0.034265198,0.0006991084,0.000025998892],"about_ca_topic_score_codex":0.0015825484,"about_ca_topic_score_gemma":0.0014681875,"teacher_disagreement_score":0.0037549536,"about_ca_system_score_codex":0.00068754447,"about_ca_system_score_gemma":0.0010223809,"threshold_uncertainty_score":0.01985836},"labels":[],"label_agreement":null},{"id":"W2123039547","doi":"10.1016/j.disc.2011.10.026","title":"Cover starters for covering arrays of strength two","year":2011,"lang":"en","type":"article","venue":"Discrete Mathematics","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University; Carleton University","funders":"","keywords":"Cover (algebra); Mathematics; Combinatorics; Heuristic; Covering space; Upper and lower bounds; Representation (politics); Discrete mathematics; Mathematical optimization; Engineering","score_opus":0.04715693982522598,"score_gpt":0.2780852931107338,"score_spread":0.23092835328550784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123039547","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17589287,0.00028940174,0.7852436,0.0005383377,0.00014989999,0.00014030312,0.00031774014,0.0031103562,0.034317493],"genre_scores_gemma":[0.7645515,0.00024262081,0.20225176,0.0004051684,0.0001291975,0.0002732207,0.00081188546,0.0014507951,0.029883761],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998502,0.00028298778,0.00007278978,0.00027877587,0.00056999444,0.00029336548],"domain_scores_gemma":[0.99331,0.004263401,0.0004350776,0.0009864792,0.00059926644,0.00040583985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009887561,0.0009944177,0.00090231077,0.0014209267,0.0011640509,0.001672255,0.001013292,0.0014950687,0.012508786],"category_scores_gemma":[0.00943219,0.0007187784,0.0009906815,0.0010463872,0.0015088732,0.0038583498,0.002791884,0.0022143333,0.002314525],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009029787,0.00019729727,0.0032871556,0.00026270034,0.000074644806,0.0010217322,0.0012420274,0.026184738,0.024102304,0.79347456,0.013559098,0.1356907],"study_design_scores_gemma":[0.00009047465,0.00020546805,0.000847258,0.00008598264,0.00007249391,0.00063126406,0.000222928,0.13654782,0.019579425,0.8283504,0.013314895,0.000051462433],"about_ca_topic_score_codex":0.00048827328,"about_ca_topic_score_gemma":0.00080023,"teacher_disagreement_score":0.012508786,"about_ca_system_score_codex":0.0007306283,"about_ca_system_score_gemma":0.00052134524,"threshold_uncertainty_score":0.041846037},"labels":[],"label_agreement":null},{"id":"W2123203095","doi":"10.1109/issre.1992.285853","title":"Control-flow based testing of Prolog programs","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Prolog; Computer science; Control flow; Selection (genetic algorithm); Programming language; Control (management); Data flow diagram; Control flow graph; Test (biology); Artificial intelligence; Database","score_opus":0.033706945543597086,"score_gpt":0.24731288896721035,"score_spread":0.21360594342361328,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123203095","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06526492,0.0004155901,0.92079294,0.00032739938,0.00005613945,0.0003627523,0.00015087705,0.0024324327,0.01019689],"genre_scores_gemma":[0.5560046,0.0002843426,0.4392343,0.00037187935,0.00006786792,0.00024461374,0.0005147267,0.00034408423,0.002933648],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955973,0.0013542444,0.00033021206,0.0003711162,0.0019992848,0.00034788947],"domain_scores_gemma":[0.9834051,0.013035401,0.00095230015,0.0006939474,0.0016969455,0.0002162943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024565426,0.0009803707,0.00062695675,0.0039821942,0.0005948421,0.0017361284,0.0012420221,0.0010750055,0.0021071576],"category_scores_gemma":[0.016939353,0.0004415285,0.0007854107,0.0014712674,0.0015347975,0.0014687763,0.00064241455,0.00068906037,0.0003316215],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078081444,0.00076625997,0.0077114017,0.0006339186,0.0001436608,0.0010957865,0.00050296815,0.2948751,0.036455918,0.120251216,0.007488363,0.5292946],"study_design_scores_gemma":[0.00006463017,0.0002876626,0.0011662403,0.00010164991,0.00004958431,0.00033198515,0.00004752541,0.91629136,0.043692835,0.033818655,0.004101843,0.000046067067],"about_ca_topic_score_codex":0.0043468373,"about_ca_topic_score_gemma":0.0043412833,"teacher_disagreement_score":0.0043468373,"about_ca_system_score_codex":0.0014478948,"about_ca_system_score_gemma":0.0008111835,"threshold_uncertainty_score":0.012991607},"labels":[],"label_agreement":null},{"id":"W2123319826","doi":"10.1007/11569596_93","title":"Generalizing Redundancy Elimination in Checking Sequences","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Sequence (biology); Redundancy (engineering); Computer science; Prime (order theory); Set (abstract data type); Finite-state machine; Algorithm; Model checking; State (computer science); Discrete mathematics; Theoretical computer science; Combinatorics; Mathematics; Programming language","score_opus":0.03110914949709723,"score_gpt":0.27836388576723836,"score_spread":0.24725473627014113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123319826","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005324724,0.00034231358,0.9860504,0.00014249614,0.00010988139,0.000078566954,0.00006580411,0.0017000537,0.006185681],"genre_scores_gemma":[0.16358507,0.0008320858,0.8199379,0.000472167,0.00029727796,0.00022911883,0.0005062029,0.001483025,0.012657186],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9957081,0.0012150003,0.00033295358,0.0008888662,0.0015140432,0.00034100882],"domain_scores_gemma":[0.99023086,0.005300561,0.0002997321,0.0032246865,0.0008493649,0.000094860276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002976614,0.0018090137,0.001362219,0.0029666093,0.0009551498,0.0014086075,0.0033942228,0.0013883335,0.00684548],"category_scores_gemma":[0.012331267,0.0016126849,0.0029700284,0.002774837,0.0031277507,0.0066821133,0.0035919042,0.0037836472,0.0023229292],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001987778,0.000102292324,0.000896622,0.00077356223,0.000094620285,0.0004913259,0.00072229735,0.040071018,0.009785899,0.5364479,0.007458762,0.402957],"study_design_scores_gemma":[0.00003756396,0.00009261046,0.00032212734,0.00020879567,0.00014664253,0.0005576871,0.00005298934,0.10737945,0.0136721525,0.856976,0.020502636,0.00005125935],"about_ca_topic_score_codex":0.002156997,"about_ca_topic_score_gemma":0.0024200687,"teacher_disagreement_score":0.00684548,"about_ca_system_score_codex":0.0010706179,"about_ca_system_score_gemma":0.0010783493,"threshold_uncertainty_score":0.022900403},"labels":[],"label_agreement":null},{"id":"W2123723258","doi":"10.1109/qsic.2009.21","title":"Web Traversal with a History Stack","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Hyperlink; Computer science; Web page; Traverse; Tree traversal; Web testing; World Wide Web; HTML5; Web navigation; Web application; Static web page; Information retrieval; Web development; Web application security; Programming language","score_opus":0.017600552813850416,"score_gpt":0.217387154459396,"score_spread":0.19978660164554557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123723258","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28061968,0.000926313,0.6231862,0.0005506003,0.00012741883,0.0004946325,0.0012754684,0.07301241,0.019807298],"genre_scores_gemma":[0.70154667,0.00037399263,0.28019178,0.00017768935,0.00003354831,0.0001793922,0.0016276786,0.0024527444,0.013416546],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99857664,0.00036223393,0.0001246913,0.00017832848,0.0005468427,0.00021128253],"domain_scores_gemma":[0.9953033,0.001554319,0.00029259053,0.00203625,0.000572243,0.00024138777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014834491,0.0006850417,0.00061334413,0.0013110895,0.0008974215,0.0015576977,0.0011142524,0.0008731742,0.005676345],"category_scores_gemma":[0.005870468,0.0006976434,0.0006622198,0.0010923237,0.0010390353,0.003837708,0.0021016533,0.000793877,0.0012617998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022688715,0.0009952502,0.036000922,0.0008432519,0.0002378553,0.0031598776,0.0018105297,0.05322793,0.071090326,0.07565911,0.020207737,0.73449826],"study_design_scores_gemma":[0.00032430407,0.001383632,0.012798552,0.00026127917,0.00039817148,0.0034563416,0.00054174406,0.6586942,0.13441896,0.09463434,0.092784606,0.0003038254],"about_ca_topic_score_codex":0.008832247,"about_ca_topic_score_gemma":0.008079709,"teacher_disagreement_score":0.008832247,"about_ca_system_score_codex":0.00065666065,"about_ca_system_score_gemma":0.0016391502,"threshold_uncertainty_score":0.018989325},"labels":[],"label_agreement":null},{"id":"W2123742395","doi":"10.1109/hase.2008.8","title":"Mutation-Based Testing of Format String Bugs","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Set (abstract data type); Mutation; Mutation testing; Exploit; Source code; Fuzz testing; String (physics); Programming language; Data mining; Software engineering; Software; Computer security","score_opus":0.05404628711613114,"score_gpt":0.2537613083169688,"score_spread":0.19971502120083767,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123742395","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6837538,0.00016072649,0.30613652,0.00022527826,0.000037700516,0.0002120558,0.0004689744,0.007400078,0.0016048665],"genre_scores_gemma":[0.86985123,0.0000715183,0.12827888,0.00010945884,0.0000097677275,0.0001424964,0.0006596744,0.00025650012,0.0006204554],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99749315,0.0007381138,0.00019173857,0.00044333757,0.0009629354,0.00017077604],"domain_scores_gemma":[0.99161255,0.005081933,0.0011725049,0.0009436061,0.001008455,0.0001809107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015460949,0.0007581768,0.00049845484,0.0017060108,0.00030098803,0.0004971381,0.0012055511,0.0007204473,0.00086173863],"category_scores_gemma":[0.012080372,0.00019908609,0.00060709263,0.00093365,0.00081890216,0.0010303034,0.0007816826,0.0004927226,0.00014934117],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010637156,0.0009097353,0.06300888,0.00055014534,0.0002493961,0.0025236767,0.00086119067,0.15082815,0.2945228,0.014773391,0.004087668,0.46662125],"study_design_scores_gemma":[0.00014451741,0.0010611324,0.016301936,0.00007339259,0.00013678354,0.0017987814,0.00017694675,0.6820872,0.28549087,0.008932691,0.0037029337,0.000092850925],"about_ca_topic_score_codex":0.0012105901,"about_ca_topic_score_gemma":0.0012405213,"teacher_disagreement_score":0.0017060108,"about_ca_system_score_codex":0.00047440862,"about_ca_system_score_gemma":0.0006863179,"threshold_uncertainty_score":0.008176684},"labels":[],"label_agreement":null},{"id":"W2124177996","doi":"10.1109/ares.2008.46","title":"Fuzzy Belief-Based Supervision","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Continuation; Formalism (music); Fuzzy logic; Fuzzy set; Artificial intelligence; Set (abstract data type); Software; Finite-state machine; Machine learning; Data mining; Algorithm; Programming language","score_opus":0.033630172601859354,"score_gpt":0.24897475747999112,"score_spread":0.21534458487813177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124177996","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036581207,0.0005459657,0.95377123,0.0003637292,0.000047178182,0.000066299566,0.00006198286,0.00059209386,0.007970233],"genre_scores_gemma":[0.8382113,0.00023409573,0.1594272,0.00007650706,0.000043796674,0.000052858755,0.0000685455,0.000023368228,0.0018624376],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99878377,0.00030822112,0.000050086343,0.00016925806,0.00059594406,0.00009275267],"domain_scores_gemma":[0.9970464,0.0016383829,0.00025979272,0.00035054548,0.0005913388,0.000113539856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014977597,0.0003991349,0.0004887268,0.0006156364,0.00048397822,0.001139877,0.00097075495,0.0006619431,0.0019594184],"category_scores_gemma":[0.0065422268,0.00022575595,0.00052742386,0.0004084981,0.0011918084,0.001716882,0.0008043733,0.0010862796,0.000314147],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007850807,0.0002737418,0.0040198606,0.00044897117,0.0001499452,0.00037942524,0.0016627826,0.1906877,0.024175197,0.3040874,0.0034895155,0.46984032],"study_design_scores_gemma":[0.00009269137,0.00021907837,0.0013625788,0.000056868328,0.000076665165,0.00017420703,0.00012115249,0.85778105,0.0134218745,0.12187934,0.004762653,0.000051869494],"about_ca_topic_score_codex":0.003604391,"about_ca_topic_score_gemma":0.0024979643,"teacher_disagreement_score":0.003604391,"about_ca_system_score_codex":0.0009777332,"about_ca_system_score_gemma":0.0009332,"threshold_uncertainty_score":0.00792098},"labels":[],"label_agreement":null},{"id":"W2124559112","doi":"10.1007/s10515-008-0043-7","title":"Parameter reference immutability: formal definition, inference tool, and comparison","year":2008,"lang":"en","type":"article","venue":"Automated Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Immutability; Correctness; Computer science; Scalability; Component (thermodynamics); Inference; Object (grammar); Programming language; Abstraction; Theoretical computer science; Artificial intelligence","score_opus":0.04144959300512138,"score_gpt":0.2640226251437893,"score_spread":0.22257303213866794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124559112","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00896859,0.00049285276,0.98430693,0.00027985708,0.00007387783,0.00008067136,0.00008844161,0.0022399267,0.0034689174],"genre_scores_gemma":[0.49485245,0.0013651574,0.49662265,0.0003487117,0.00031522472,0.00028372186,0.00069006876,0.0019813052,0.0035407022],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.982406,0.0047135805,0.002080487,0.0023730618,0.007135832,0.0012909413],"domain_scores_gemma":[0.93717086,0.033967007,0.002868006,0.01940528,0.006044341,0.00054456474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0146724135,0.0015660945,0.0021480354,0.0073326533,0.001707279,0.00573528,0.007921661,0.0036636824,0.006360824],"category_scores_gemma":[0.061756853,0.0016473258,0.0025767768,0.005495082,0.007989756,0.018831147,0.005295845,0.005375048,0.0013571765],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034289577,0.00026805076,0.00389579,0.00042045693,0.00008600703,0.00038848436,0.0007547575,0.015016187,0.0049902177,0.8154135,0.0026435384,0.15577999],"study_design_scores_gemma":[0.00012586053,0.00029801516,0.0011160832,0.0003577072,0.0003700573,0.0019916508,0.0003554036,0.22155955,0.042732734,0.7118653,0.01900513,0.00022239264],"about_ca_topic_score_codex":0.001681558,"about_ca_topic_score_gemma":0.0011905794,"teacher_disagreement_score":0.0146724135,"about_ca_system_score_codex":0.0020397885,"about_ca_system_score_gemma":0.0031276615,"threshold_uncertainty_score":0.07759607},"labels":[],"label_agreement":null},{"id":"W2124756729","doi":"10.1007/978-3-540-30482-1_12","title":"Formal Models for Web Navigations with Session Control and Browser Cache","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Session (web analytics); Web modeling; Web application security; World Wide Web; Web page; Web service; Web navigation; Web API; Static web page; Cache; Web development; Operating system","score_opus":0.018218756490812737,"score_gpt":0.249842034535426,"score_spread":0.23162327804461327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124756729","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0108181685,0.0005928342,0.97096205,0.00090310554,0.00010138078,0.00010463597,0.000382535,0.0010167242,0.015118606],"genre_scores_gemma":[0.52661407,0.0014559586,0.44066486,0.00055304036,0.0002987346,0.00090009556,0.001349402,0.0008413803,0.02732231],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9977252,0.00078081485,0.00025197936,0.0003135578,0.0006142085,0.00031417495],"domain_scores_gemma":[0.9958406,0.0022961313,0.00035636,0.0007746886,0.00056875177,0.00016337444],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022574156,0.0013481753,0.00078058685,0.0011492305,0.0013234522,0.0044560665,0.0027702278,0.0027239758,0.006514085],"category_scores_gemma":[0.006315905,0.0015246405,0.002149022,0.0012120168,0.0043659084,0.007157467,0.0016731931,0.004589544,0.0016891194],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019459761,0.000042107265,0.00014483536,0.000058822035,0.000012166648,0.00007343966,0.00033811704,0.0239673,0.0005984583,0.96898973,0.0009779059,0.0047776597],"study_design_scores_gemma":[0.00003950228,0.00001941416,0.00008230078,0.00005765533,0.000032556778,0.00008450943,0.00009801601,0.10203111,0.00089921604,0.885279,0.011349592,0.000027156433],"about_ca_topic_score_codex":0.011841503,"about_ca_topic_score_gemma":0.011935845,"teacher_disagreement_score":0.011841503,"about_ca_system_score_codex":0.0033935437,"about_ca_system_score_gemma":0.002522966,"threshold_uncertainty_score":0.024622023},"labels":[],"label_agreement":null},{"id":"W2125172218","doi":"10.1109/infcom.1996.493058","title":"Context independent unique sequences generation for protocol testing","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Bell (Canada)","funders":"","keywords":"Computer science; Context (archaeology); Protocol (science); Geology; Paleontology","score_opus":0.16702740374403804,"score_gpt":0.31924741069067436,"score_spread":0.15222000694663632,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2125172218","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00912541,0.00013491588,0.98762137,0.00006155294,0.000029859506,0.00013450054,0.000043510052,0.0015403469,0.0013085146],"genre_scores_gemma":[0.17707072,0.00015819739,0.8205539,0.00011128508,0.00004237053,0.0004504916,0.00028994313,0.00027950836,0.0010435618],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99560297,0.0020941258,0.00027878306,0.00053729775,0.0012586504,0.0002281271],"domain_scores_gemma":[0.9931924,0.0038277425,0.00051109237,0.0014220927,0.00089628785,0.00015040667],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021805272,0.0009948853,0.0006844123,0.0017679308,0.0007085007,0.000988176,0.0013985303,0.0009932338,0.002336496],"category_scores_gemma":[0.012371637,0.0004123535,0.00076821167,0.0012267395,0.0014066697,0.0020487478,0.001482906,0.0015788865,0.0005990106],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004417782,0.00014374769,0.002222251,0.00041456916,0.000052054515,0.0004370944,0.0003469131,0.11155819,0.02216304,0.3544187,0.0037821454,0.5040195],"study_design_scores_gemma":[0.000062047286,0.0002294418,0.00041645597,0.00011850411,0.0000421444,0.000437059,0.00006104061,0.71796376,0.04603337,0.22222772,0.012351385,0.000057100275],"about_ca_topic_score_codex":0.00060825795,"about_ca_topic_score_gemma":0.0007988837,"teacher_disagreement_score":0.002336496,"about_ca_system_score_codex":0.0009068378,"about_ca_system_score_gemma":0.0013484412,"threshold_uncertainty_score":0.011531889},"labels":[],"label_agreement":null},{"id":"W2125389711","doi":"10.1109/compsac.2007.11","title":"A Combined Concept Location Method for Java Programs","year":2007,"lang":"en","type":"article","venue":"Proceedings - International Computer Software & Applications Conference","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Algoma University; Laurentian University","funders":"","keywords":"Debugging; Computer science; Tracing; Java; Profiling (computer programming); Graph; Call graph; Static analysis; Programming language; Software; Source code; Theoretical computer science; Distributed computing","score_opus":0.0300423036523799,"score_gpt":0.31690703169339296,"score_spread":0.28686472804101304,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2125389711","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031380272,0.000075824,0.9882604,0.000060614962,0.000026138856,0.0000654084,0.00006373474,0.007675205,0.0006346895],"genre_scores_gemma":[0.059313867,0.000080982754,0.9367282,0.000064536696,0.000025180525,0.00019679549,0.00016818514,0.00070291397,0.0027193879],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9980989,0.00042438554,0.0001782673,0.0004368195,0.0007688075,0.00009282139],"domain_scores_gemma":[0.99725395,0.0011268936,0.0002638059,0.00063329516,0.0006081617,0.00011390581],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018492733,0.00086706225,0.0007029116,0.0023524829,0.0005301425,0.0014078203,0.0020080293,0.0013502956,0.0056603886],"category_scores_gemma":[0.006691612,0.00065312424,0.00075264816,0.0013196403,0.0008851459,0.0034414728,0.0020850056,0.0011755717,0.0017780268],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003742808,0.00015661528,0.0014207005,0.00043303677,0.00006821844,0.00040708738,0.00070781435,0.0065982128,0.03805923,0.03515101,0.007095171,0.90952855],"study_design_scores_gemma":[0.000625304,0.0008135891,0.0021855978,0.00030346465,0.00023855292,0.003930926,0.00032488056,0.5986018,0.15900214,0.05843548,0.17517537,0.00036288804],"about_ca_topic_score_codex":0.0010566124,"about_ca_topic_score_gemma":0.001187311,"teacher_disagreement_score":0.0056603886,"about_ca_system_score_codex":0.00049534,"about_ca_system_score_gemma":0.0010233778,"threshold_uncertainty_score":0.01893586},"labels":[],"label_agreement":null},{"id":"W2125596330","doi":"10.1109/issre.2005.28","title":"Minimization of Randomized Unit Test Cases","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":68,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Test (biology); Computer science; Minification; Unit (ring theory); Mathematics; Geology; World Wide Web","score_opus":0.019213582231454827,"score_gpt":0.2543931640386862,"score_spread":0.2351795818072314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2125596330","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011812532,0.000115235205,0.9856554,0.00020361836,0.000014808867,0.00024373758,0.00006558319,0.00060717366,0.0012818677],"genre_scores_gemma":[0.24573816,0.00012524343,0.7510348,0.00024061659,0.000048127215,0.0014049928,0.00032587885,0.00026065734,0.0008214804],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9762523,0.0147741195,0.0010190452,0.0018958063,0.005275043,0.00078365585],"domain_scores_gemma":[0.91949487,0.06460686,0.0043763183,0.0074420855,0.003436189,0.0006436753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011350793,0.0013401832,0.0014902261,0.0021929813,0.0005876757,0.0012783301,0.003371496,0.0013564657,0.003254118],"category_scores_gemma":[0.081492305,0.0007164102,0.0011348267,0.0015664689,0.002516992,0.0022823557,0.0020347985,0.0021510546,0.0006929005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005552912,0.00066248863,0.0060006496,0.00055376237,0.0002093341,0.00042220685,0.00024964492,0.49177203,0.016615715,0.18306464,0.004525683,0.29536858],"study_design_scores_gemma":[0.00017381937,0.00056305714,0.0007999909,0.00006877342,0.000062864354,0.00028146457,0.000035557438,0.8754546,0.008698465,0.11096011,0.0028618153,0.00003947932],"about_ca_topic_score_codex":0.0011353598,"about_ca_topic_score_gemma":0.0012217747,"teacher_disagreement_score":0.011350793,"about_ca_system_score_codex":0.0013106613,"about_ca_system_score_gemma":0.00217373,"threshold_uncertainty_score":0.060029447},"labels":[],"label_agreement":null},{"id":"W2125727889","doi":"10.1145/1368088.1368136","title":"Sufficient mutation operators for measuring test effectiveness","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":181,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Mutant; Mutation testing; Test suite; Mutation; Measure (data warehouse); Computer science; Process (computing); Mathematics; Test case; Data mining; Genetics; Biology; Machine learning; Programming language; Gene","score_opus":0.04459589575011752,"score_gpt":0.26666301761281497,"score_spread":0.22206712186269745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2125727889","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37934908,0.0036152424,0.6006094,0.0004897168,0.00014559877,0.0006364717,0.0027813627,0.0035315594,0.008841516],"genre_scores_gemma":[0.85748124,0.0005051683,0.13791199,0.00019371348,0.00015471282,0.0007030097,0.0020735592,0.00044150563,0.0005349718],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9551332,0.013485865,0.0048392755,0.0033903492,0.021748705,0.0014026103],"domain_scores_gemma":[0.8653868,0.10073735,0.012469004,0.009414095,0.0099447025,0.0020480321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013332378,0.0020051426,0.0019978029,0.007645034,0.0006736141,0.0014925309,0.0018635774,0.0029031967,0.0027939808],"category_scores_gemma":[0.08885669,0.00054213434,0.001333758,0.0028306702,0.0022237909,0.0040028933,0.0013326305,0.0013887157,0.00096149085],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038303405,0.0017766524,0.13107379,0.0026350047,0.0010277745,0.001794962,0.0008616951,0.19959366,0.19446316,0.12291531,0.0073161637,0.33271155],"study_design_scores_gemma":[0.00045082276,0.0027973182,0.06105242,0.00053889514,0.0005194231,0.005045681,0.0003257138,0.6557104,0.14634727,0.11590138,0.011000593,0.00031011022],"about_ca_topic_score_codex":0.00069132406,"about_ca_topic_score_gemma":0.00065430143,"teacher_disagreement_score":0.013332378,"about_ca_system_score_codex":0.0010205236,"about_ca_system_score_gemma":0.0011903516,"threshold_uncertainty_score":0.070509195},"labels":[],"label_agreement":null},{"id":"W2126024083","doi":"10.1109/pacrim.1995.519425","title":"Hardware techniques for testing software components","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Class (philosophy); Software performance testing; Scheme (mathematics); Software; Software engineering; Object-oriented programming; Unit testing; Software testing; System integration testing; Programming language; Component-based software engineering; Software construction; Embedded system; Software system; Artificial intelligence","score_opus":0.0930962594380544,"score_gpt":0.27308306823971595,"score_spread":0.17998680880166157,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126024083","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052267946,0.0003826409,0.9878647,0.00010924261,0.000047397443,0.00012064639,0.000043001633,0.0017173617,0.0044882423],"genre_scores_gemma":[0.1055372,0.00075507065,0.8887007,0.00012603258,0.000066717694,0.00049803423,0.00019936774,0.00028314907,0.0038336213],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99602413,0.0009073205,0.0003244553,0.0003824648,0.0021350805,0.00022652315],"domain_scores_gemma":[0.9935068,0.0031993529,0.00052859524,0.0019377604,0.000710646,0.00011689021],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016144003,0.0012702129,0.0006237259,0.0022901904,0.00078162266,0.0011411068,0.0027033314,0.0009866755,0.009032955],"category_scores_gemma":[0.0069256257,0.0007188239,0.0007373731,0.0023208128,0.0017476535,0.0026130814,0.0017483808,0.0019925684,0.001901064],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023815996,0.00023590194,0.0022909376,0.0011033303,0.00009618542,0.00032278526,0.00065398467,0.015802378,0.11324893,0.20210172,0.0051979246,0.6587078],"study_design_scores_gemma":[0.00045114278,0.0018841238,0.0044866134,0.0007080449,0.0003876929,0.0024841812,0.0004606251,0.15170929,0.42103446,0.2806167,0.13556817,0.00020898893],"about_ca_topic_score_codex":0.00072797097,"about_ca_topic_score_gemma":0.00111679,"teacher_disagreement_score":0.009032955,"about_ca_system_score_codex":0.00058636256,"about_ca_system_score_gemma":0.0008973129,"threshold_uncertainty_score":0.030218303},"labels":[],"label_agreement":null},{"id":"W2128209135","doi":"10.1109/sefm.2009.12","title":"Checking Sequence Construction Using Adaptive and Preset Distinguishing Sequences","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Sabancı Üniversitesi","keywords":"Sequence (biology); Computer science; Model checking; Algorithm; State (computer science); Finite-state machine; Theoretical computer science; Biology","score_opus":0.07917394972292258,"score_gpt":0.31378634299328745,"score_spread":0.23461239327036487,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128209135","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22611411,0.00014563209,0.770132,0.000056257715,0.000022706425,0.00027061382,0.00012839185,0.0023823334,0.0007480508],"genre_scores_gemma":[0.46695876,0.000071283655,0.531589,0.000037593407,0.000011952865,0.00022534921,0.0003532017,0.00026555703,0.00048736462],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953862,0.0013083846,0.0006568702,0.0010386865,0.0013931588,0.000216663],"domain_scores_gemma":[0.95301646,0.029262325,0.0048437845,0.009525956,0.0027670844,0.00058441464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036007343,0.00073145906,0.00048067438,0.0014628642,0.00040805925,0.0007014728,0.00142037,0.0007893024,0.001675463],"category_scores_gemma":[0.019447934,0.0005446593,0.00053084694,0.00088383415,0.0012244781,0.0022322077,0.0011802635,0.0015797331,0.00035348948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030719447,0.0008117652,0.01515364,0.0010467292,0.00013299193,0.0005813393,0.0010750329,0.07521901,0.31209525,0.040959526,0.0007846693,0.54906815],"study_design_scores_gemma":[0.00026626827,0.0027547232,0.0050514364,0.000116668125,0.0001421289,0.0012784858,0.0001710745,0.32854953,0.63137996,0.024453994,0.00567026,0.00016549071],"about_ca_topic_score_codex":0.00029249952,"about_ca_topic_score_gemma":0.0005010594,"teacher_disagreement_score":0.0036007343,"about_ca_system_score_codex":0.0005267991,"about_ca_system_score_gemma":0.00096416124,"threshold_uncertainty_score":0.01904273},"labels":[],"label_agreement":null},{"id":"W2128416870","doi":"10.1007/s00165-005-0083-8","title":"Constructing checking sequences for distributed testing","year":2006,"lang":"en","type":"article","venue":"Formal Aspects of Computing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Observability; Controllability; Computer science; Sequence (biology); Finite-state machine; Theory of computation; Synchronization (alternating current); Automaton; Model checking; Implementation; State (computer science); Sequence diagram; Observable; Reset (finance); Distributed computing; Programming language; Theoretical computer science; Unified Modeling Language; Mathematics; Software","score_opus":0.023429877115581837,"score_gpt":0.25895384582965925,"score_spread":0.2355239687140774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128416870","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041369542,0.00004627491,0.9549347,0.000072916206,0.000031935644,0.00018220075,0.00005735738,0.001999913,0.0013050266],"genre_scores_gemma":[0.3429168,0.000051915453,0.6547126,0.000057839807,0.000026173253,0.0003027858,0.00035026457,0.00023535518,0.0013462756],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9962251,0.0014309997,0.00034886991,0.00065446773,0.0010929278,0.00024760625],"domain_scores_gemma":[0.9810404,0.011875452,0.0013709742,0.0030743412,0.0021899834,0.00044879006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035996775,0.00075766357,0.00064515043,0.00143991,0.00084721047,0.0009056743,0.001210687,0.0009583494,0.0029121202],"category_scores_gemma":[0.013913869,0.00059053727,0.00074274844,0.000706617,0.0019139807,0.0016380161,0.0016123158,0.0015799527,0.00065721286],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011253239,0.00059625995,0.0067106183,0.0006416083,0.00007510799,0.0015298297,0.001712985,0.17560638,0.06326568,0.2819869,0.0029301485,0.4638192],"study_design_scores_gemma":[0.00019409077,0.00072053145,0.00090640783,0.00022237489,0.00007876491,0.0006147409,0.00021376183,0.6400704,0.11652376,0.22518247,0.015198302,0.0000744734],"about_ca_topic_score_codex":0.0009892602,"about_ca_topic_score_gemma":0.00093741465,"teacher_disagreement_score":0.0035996775,"about_ca_system_score_codex":0.00077218306,"about_ca_system_score_gemma":0.0016304114,"threshold_uncertainty_score":0.019037127},"labels":[],"label_agreement":null},{"id":"W2128952545","doi":"10.1109/csmr.2002.995791","title":"A generic worklist algorithm for graph reachability problems in program analysis","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reachability; Computer science; Graph; Reachability problem; Theoretical computer science; Algorithm","score_opus":0.030170705854607422,"score_gpt":0.28858534191547247,"score_spread":0.25841463606086507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128952545","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010481764,0.0000534299,0.9938678,0.00010895469,0.000014789128,0.00010336066,0.00007409007,0.0031205048,0.0016090125],"genre_scores_gemma":[0.017353218,0.00012616033,0.9796963,0.000098296696,0.0000291623,0.00023764661,0.00035617736,0.0007005263,0.0014025545],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99790823,0.0005270228,0.00023243758,0.00046595887,0.0006717568,0.00019456091],"domain_scores_gemma":[0.99724126,0.0015244501,0.00019211658,0.00063358975,0.0003297391,0.000078848396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027798396,0.0021356866,0.0011049575,0.0028738913,0.0015649452,0.003069599,0.003791926,0.0028409222,0.010466641],"category_scores_gemma":[0.00808129,0.00093148777,0.002692518,0.0026710692,0.0022453633,0.00571203,0.0034797778,0.002693548,0.0040592565],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024005669,0.0003506632,0.0011717072,0.0009143828,0.00010959186,0.00033232747,0.0005650122,0.07226126,0.01428723,0.3190836,0.018878927,0.5718053],"study_design_scores_gemma":[0.00022270533,0.00016025879,0.0002800916,0.000254746,0.00015203515,0.00054669013,0.00021175778,0.4250307,0.022739867,0.50015205,0.050139107,0.00011000925],"about_ca_topic_score_codex":0.0018947202,"about_ca_topic_score_gemma":0.0033612507,"teacher_disagreement_score":0.010466641,"about_ca_system_score_codex":0.0016627802,"about_ca_system_score_gemma":0.0022677968,"threshold_uncertainty_score":0.03501439},"labels":[],"label_agreement":null},{"id":"W2129272701","doi":"10.5539/cis.v3n1p35","title":"A New Method of Reducing Pair-wise Combinatorial Test Suite","year":2010,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Test suite; Computer science; Combinatorial design; Test (biology); Combinatorial explosion; Combinatorial optimization; Test case; Algorithm; Suite; Combinatorial method; Mathematics; Combinatorics; Machine learning","score_opus":0.012833545615482174,"score_gpt":0.2873989325205463,"score_spread":0.27456538690506416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129272701","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006131,0.00015566386,0.990217,0.00012297923,0.000051906685,0.00016681702,0.00006835354,0.0010740083,0.002012395],"genre_scores_gemma":[0.08914004,0.00018589366,0.90651894,0.00019717467,0.00007770596,0.00053893327,0.0004916364,0.00036186856,0.0024879547],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99631065,0.0007755844,0.00018350345,0.00067369075,0.0018856835,0.00017086959],"domain_scores_gemma":[0.9969799,0.0012405583,0.0002574477,0.00061070983,0.0008050779,0.0001063555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011700038,0.0016315628,0.0013771915,0.0038419636,0.0009137427,0.0010978642,0.0020925375,0.00097214826,0.0034758118],"category_scores_gemma":[0.0060954643,0.0006370379,0.0018865902,0.0022759736,0.0010559406,0.0017848127,0.0015438338,0.0019969502,0.0008238696],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017510132,0.00040966997,0.0026628652,0.0005315435,0.00020927365,0.0004990533,0.00026991754,0.06347152,0.052270856,0.038771816,0.008728469,0.83199996],"study_design_scores_gemma":[0.00021879452,0.00091007416,0.0033109502,0.00010441184,0.00042322924,0.0031004539,0.00012657483,0.8132527,0.08044976,0.056427665,0.04150721,0.00016821953],"about_ca_topic_score_codex":0.0016078347,"about_ca_topic_score_gemma":0.0017673547,"teacher_disagreement_score":0.0038419636,"about_ca_system_score_codex":0.00074109546,"about_ca_system_score_gemma":0.0016994437,"threshold_uncertainty_score":0.011627734},"labels":[],"label_agreement":null},{"id":"W2129718707","doi":"10.1016/j.infsof.2006.11.002","title":"A state-based approach to integration testing based on UML models","year":2006,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":103,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Unified Modeling Language; Computer science; Integration testing; Test case; Class diagram; Reliability engineering; Sequence diagram; Data mining; Software; Software engineering; Programming language; Engineering; Machine learning","score_opus":0.01904630375427704,"score_gpt":0.22558055522803536,"score_spread":0.2065342514737583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129718707","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003976123,0.000054803913,0.99240875,0.00012032343,0.000016935755,0.000053014028,0.000032397682,0.0016563237,0.0016812833],"genre_scores_gemma":[0.29039916,0.00020865473,0.705842,0.0001386617,0.000032820124,0.00028284124,0.00020519074,0.0003890171,0.002501593],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9963638,0.0013102331,0.00023136451,0.0003521442,0.0015714553,0.00017109004],"domain_scores_gemma":[0.99300426,0.0037553825,0.00038102965,0.0017155968,0.0010155516,0.000128221],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028570245,0.0009029382,0.0008676383,0.0020574175,0.00081318215,0.002984883,0.0025424433,0.0016456572,0.0034253793],"category_scores_gemma":[0.010069295,0.0011140498,0.0016116074,0.001102881,0.0018968442,0.0051734266,0.0019514582,0.0023856086,0.00066071504],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029665054,0.00050334726,0.0023002597,0.00028741854,0.00019764842,0.00039775236,0.0012500703,0.24562086,0.018949907,0.4850573,0.0031513602,0.2419875],"study_design_scores_gemma":[0.000030852923,0.000076026954,0.00021752685,0.000067909685,0.000101105754,0.000105459,0.000039638653,0.8896341,0.007501062,0.09858515,0.0036022512,0.000038985505],"about_ca_topic_score_codex":0.004627174,"about_ca_topic_score_gemma":0.00678359,"teacher_disagreement_score":0.004627174,"about_ca_system_score_codex":0.0012937431,"about_ca_system_score_gemma":0.0016787995,"threshold_uncertainty_score":0.015109599},"labels":[],"label_agreement":null},{"id":"W2129772051","doi":"10.1109/ares.2007.49","title":"Automatic Failure Detection with Separation of Concerns","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Session (web analytics); Rotation formalisms in three dimensions; Finite-state machine; Separation of concerns; State (computer science); Software; Reliability engineering; Distributed computing; Programming language; Engineering","score_opus":0.015875870747886055,"score_gpt":0.288597716786828,"score_spread":0.27272184603894195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129772051","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008707093,0.00011242247,0.9866788,0.000073619485,0.00001717695,0.00005174312,0.000026527414,0.0038856652,0.00044686574],"genre_scores_gemma":[0.37003994,0.00015071323,0.62707454,0.0001414292,0.00004910402,0.0002597974,0.00019251218,0.0008707897,0.0012212517],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9905725,0.0027934408,0.0007151805,0.0014368654,0.00394354,0.0005384809],"domain_scores_gemma":[0.98060304,0.009818395,0.0019928087,0.0053050234,0.002055626,0.00022505209],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004715276,0.001691936,0.0014559161,0.0020288455,0.0005569288,0.0021370812,0.0025753116,0.001498045,0.0017178351],"category_scores_gemma":[0.020076286,0.0009547522,0.0012863998,0.000934139,0.001474508,0.003739067,0.003531393,0.0029527799,0.0008961565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015213862,0.00034022794,0.0060882694,0.000899896,0.00035320496,0.0012041262,0.0015192814,0.05883267,0.1296591,0.0911373,0.004964653,0.7034799],"study_design_scores_gemma":[0.00019617147,0.00037714723,0.0012246915,0.00013158364,0.0001600108,0.0011665762,0.00013777176,0.7273743,0.14059104,0.1163156,0.01215107,0.00017404044],"about_ca_topic_score_codex":0.00066402636,"about_ca_topic_score_gemma":0.00045325138,"teacher_disagreement_score":0.004715276,"about_ca_system_score_codex":0.00057988387,"about_ca_system_score_gemma":0.0011681502,"threshold_uncertainty_score":0.024937093},"labels":[],"label_agreement":null},{"id":"W2130262632","doi":"10.1016/j.tcs.2009.07.057","title":"Covering arrays avoiding forbidden edges","year":2009,"lang":"en","type":"article","venue":"Theoretical Computer Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; Carleton University; University of Toronto; Toronto Metropolitan University","funders":"","keywords":"Pairwise comparison; Clique; Generalization; Enhanced Data Rates for GSM Evolution; Binary number; Software; Mathematics; Computer science; Algorithm; Theoretical computer science; Combinatorics; Discrete mathematics; Artificial intelligence; Arithmetic; Programming language","score_opus":0.01674969832501048,"score_gpt":0.26296799658636427,"score_spread":0.2462182982613538,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2130262632","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31003392,0.00038254322,0.63612545,0.0007752254,0.00019687225,0.00010318934,0.00062408374,0.0038272422,0.04793149],"genre_scores_gemma":[0.81257164,0.00027828265,0.17239638,0.00036143608,0.00006723438,0.00016432627,0.00095752033,0.0007391915,0.012463928],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99876356,0.0003298125,0.00009034139,0.00021155922,0.00036039914,0.00024429636],"domain_scores_gemma":[0.99128574,0.0048233345,0.00064225745,0.0019040251,0.0009982613,0.0003462691],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00053728867,0.0006558079,0.0007565795,0.0012221291,0.0011296791,0.0016523886,0.0011604395,0.0010421263,0.006387398],"category_scores_gemma":[0.00629546,0.0007440818,0.0006000223,0.0019094696,0.0008841004,0.0029288074,0.0019206911,0.001485769,0.0014569146],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00087806483,0.0002279877,0.004133053,0.00041005085,0.00008873368,0.0010144578,0.00082876073,0.034707844,0.050110467,0.6688594,0.013296973,0.22544427],"study_design_scores_gemma":[0.000087515335,0.00018130676,0.0010287747,0.00008581486,0.00009575778,0.001033831,0.00028296458,0.09181452,0.038668785,0.8444489,0.022220246,0.00005157052],"about_ca_topic_score_codex":0.0004255806,"about_ca_topic_score_gemma":0.0006170245,"teacher_disagreement_score":0.006387398,"about_ca_system_score_codex":0.00045232833,"about_ca_system_score_gemma":0.00047075102,"threshold_uncertainty_score":0.021368027},"labels":[],"label_agreement":null},{"id":"W2131009004","doi":"10.1002/spe.452","title":"A framework for table driven testing of Java classes","year":2002,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Unit testing; Java; Software portability; Tuple; Class (philosophy); Table (database); Programming language; Integration testing; Interface (matter); Standardization; Software engineering; Class hierarchy; Reuse; Object-oriented programming; Data mining; Operating system; Artificial intelligence; Software; Engineering","score_opus":0.06962558702449549,"score_gpt":0.33178758210427484,"score_spread":0.26216199507977933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2131009004","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015201241,0.00009219826,0.9884802,0.0000856334,0.000020832047,0.00015354386,0.000084062274,0.008691867,0.00087154936],"genre_scores_gemma":[0.05885888,0.00016305529,0.9375095,0.000111791065,0.00003551068,0.00043700158,0.0005030078,0.0012130061,0.0011682601],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9935967,0.0021967,0.00064998755,0.00088107056,0.0023042366,0.00037122695],"domain_scores_gemma":[0.99109757,0.005203645,0.00059962826,0.0017520505,0.0010543669,0.0002928001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008710796,0.0011855271,0.0011315811,0.0024535288,0.0007352319,0.0037343556,0.0047042244,0.0017929944,0.0053165285],"category_scores_gemma":[0.01594419,0.001253704,0.0024274543,0.0009930851,0.002427772,0.0035597817,0.0022037548,0.0024123301,0.0014295882],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028122347,0.00032103594,0.0030466842,0.000824962,0.00019810637,0.0014525488,0.0012684033,0.2132815,0.016694756,0.37351754,0.01241464,0.37669858],"study_design_scores_gemma":[0.000105868705,0.0001729066,0.00048342047,0.00032746978,0.00006364781,0.00076500006,0.00009021967,0.77076757,0.012960126,0.15716328,0.057000037,0.00010044786],"about_ca_topic_score_codex":0.00508022,"about_ca_topic_score_gemma":0.003999047,"teacher_disagreement_score":0.008710796,"about_ca_system_score_codex":0.0015803387,"about_ca_system_score_gemma":0.0018850035,"threshold_uncertainty_score":0.046067655},"labels":[],"label_agreement":null},{"id":"W2131288453","doi":"10.1016/s0140-3664(00)00229-2","title":"Rapid generation of functional tests using MSCs, SDL and TTCN","year":2001,"lang":"en","type":"article","venue":"Computer Communications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science","score_opus":0.21870477160732624,"score_gpt":0.32048971299808116,"score_spread":0.10178494139075492,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2131288453","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04134282,0.00009779946,0.93108094,0.000111629146,0.0000940295,0.0002131584,0.00036207738,0.023439893,0.0032576714],"genre_scores_gemma":[0.51065314,0.0000983586,0.48195863,0.00013579676,0.000040444127,0.00034582545,0.0013083718,0.0027648734,0.0026945504],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99649054,0.0010378787,0.00026648142,0.00035757094,0.0015780396,0.00026953727],"domain_scores_gemma":[0.98608726,0.008347346,0.0008913479,0.0018166042,0.00254081,0.0003165413],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023415799,0.001250886,0.0007893126,0.0022007506,0.00031390152,0.0011012339,0.0015548252,0.0007687935,0.005127454],"category_scores_gemma":[0.0142726395,0.00047383853,0.0007368062,0.0007150239,0.0008382222,0.0012827324,0.0013143208,0.00088243524,0.0012966946],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00177952,0.0005225426,0.00697836,0.0007658639,0.00011674377,0.001355238,0.00055197976,0.08207212,0.15741023,0.030809678,0.010587293,0.7070505],"study_design_scores_gemma":[0.00033906524,0.0007376584,0.0011392176,0.00011792206,0.00009799072,0.000553636,0.000117250696,0.6877318,0.27627572,0.019476691,0.013345956,0.000067153385],"about_ca_topic_score_codex":0.0017022161,"about_ca_topic_score_gemma":0.0021453593,"teacher_disagreement_score":0.005127454,"about_ca_system_score_codex":0.00063016,"about_ca_system_score_gemma":0.0012851573,"threshold_uncertainty_score":0.017153084},"labels":[],"label_agreement":null},{"id":"W2132123383","doi":"10.1109/ccece.2004.1345041","title":"Incorporating a contract-based test facility to the GUI framework","year":2004,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Graphical user interface; Graphical user interface testing; Reuse; Software engineering; Testability; Source code; Object-oriented programming; Software; Code coverage; Code reuse; Test case; User interface; Programming language; Reliability engineering; Engineering; User interface design","score_opus":0.022751505782270087,"score_gpt":0.27067079836287283,"score_spread":0.24791929258060275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132123383","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014004385,0.000113082635,0.97693956,0.00051870727,0.000058351237,0.00029072215,0.000030123878,0.0035426042,0.0045024813],"genre_scores_gemma":[0.2788674,0.00014707213,0.7170063,0.0004867126,0.000086359534,0.0003424082,0.00013161552,0.0006414472,0.0022906472],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99038565,0.0034523373,0.0008372647,0.0007595139,0.0037478288,0.00081740745],"domain_scores_gemma":[0.9796219,0.008399539,0.0020097368,0.005383662,0.0035691552,0.0010159314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010556638,0.0006626434,0.0006333895,0.0013372692,0.0008487376,0.0021000188,0.0038594732,0.0022914982,0.002080138],"category_scores_gemma":[0.023510285,0.00068096374,0.0009917397,0.0008030426,0.002885144,0.0036857298,0.0031185788,0.0034833846,0.0006491372],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008085559,0.0013392807,0.009646652,0.000638228,0.0001441553,0.0030225415,0.00174139,0.07103221,0.08264231,0.4379153,0.0085416185,0.3825278],"study_design_scores_gemma":[0.0004584335,0.0013731134,0.0034659388,0.00040243857,0.000202159,0.0040672384,0.0002577629,0.66420525,0.1105962,0.087119326,0.12750193,0.00035021701],"about_ca_topic_score_codex":0.0037411582,"about_ca_topic_score_gemma":0.0019312358,"teacher_disagreement_score":0.010556638,"about_ca_system_score_codex":0.0013573298,"about_ca_system_score_gemma":0.004026496,"threshold_uncertainty_score":0.055829525},"labels":[],"label_agreement":null},{"id":"W2132409828","doi":"10.1145/2597809.2597823","title":"em-SPADE","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Compiler; Bridge (graph theory); Programming language; Parallel computing; Computer architecture; Embedded system","score_opus":0.011332472989620519,"score_gpt":0.2311082114029438,"score_spread":0.2197757384133233,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132409828","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029703101,0.00073737523,0.60178554,0.0007722131,0.0006643495,0.00030718299,0.00690725,0.25903916,0.10008378],"genre_scores_gemma":[0.2752775,0.0007221474,0.5430148,0.002056365,0.00021396347,0.0007068284,0.026884405,0.03782393,0.11330007],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99908566,0.00014691461,0.00006381451,0.00022892446,0.0003743001,0.00010035258],"domain_scores_gemma":[0.9982494,0.00040490116,0.00009373856,0.0007611083,0.0004322417,0.00005865198],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008775268,0.0008412027,0.00055622857,0.00076568656,0.00045378786,0.0013071545,0.0018350895,0.0008043162,0.034633454],"category_scores_gemma":[0.0038344413,0.00075045304,0.0007181602,0.00046616077,0.000565332,0.0027451538,0.0019673028,0.0015467862,0.019756602],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001688147,0.0004292431,0.0052616126,0.0008008763,0.00018828106,0.0005104359,0.00035805703,0.019636404,0.027717073,0.08928554,0.39352915,0.46059528],"study_design_scores_gemma":[0.0003671502,0.00037789322,0.0015947061,0.0001354679,0.00009945222,0.001057861,0.00007364496,0.17571342,0.123795606,0.050769154,0.6459316,0.000084037354],"about_ca_topic_score_codex":0.00064276083,"about_ca_topic_score_gemma":0.0010363676,"teacher_disagreement_score":0.034633454,"about_ca_system_score_codex":0.000384435,"about_ca_system_score_gemma":0.0010406917,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2132430857","doi":"10.1109/icsm.2004.1357874","title":"Context driven slicing based coupling measures","year":2004,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Program slicing; Slicing; Computer science; Coupling (piping); Context (archaeology); Source code; Software quality; Programming language; Software; Object-oriented programming; Code (set theory); Distributed computing; Software development; Engineering; Computer graphics (images)","score_opus":0.03167786515567414,"score_gpt":0.2563372031324804,"score_spread":0.22465933797680623,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132430857","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09894608,0.0005766542,0.89487517,0.00004784926,0.000031561285,0.000185568,0.0003826052,0.0028867987,0.0020676784],"genre_scores_gemma":[0.67833024,0.00019256696,0.32002234,0.000038221955,0.00003169533,0.00032319897,0.0005263469,0.00021189215,0.00032353066],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9934068,0.0012386855,0.0005816988,0.0006914597,0.003760703,0.00032072084],"domain_scores_gemma":[0.9897428,0.0027208074,0.0027034974,0.0017785024,0.0026670056,0.00038736354],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003185114,0.0014379834,0.0007731911,0.005148242,0.0005850331,0.0014416957,0.0013288086,0.0007142782,0.0009121004],"category_scores_gemma":[0.011789209,0.0003800655,0.00063157565,0.0025243473,0.0011456012,0.001926302,0.0011262225,0.0009498178,0.00016035733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078460405,0.0004250525,0.05881514,0.00087635074,0.00045372304,0.00044571597,0.0012302446,0.25244826,0.19830585,0.09157472,0.0019826996,0.3926577],"study_design_scores_gemma":[0.000057979083,0.0014311578,0.031293463,0.00013758578,0.00021313461,0.00058086065,0.00024893094,0.76780236,0.14816305,0.044188667,0.0056218524,0.000260965],"about_ca_topic_score_codex":0.0021632048,"about_ca_topic_score_gemma":0.0018736627,"teacher_disagreement_score":0.005148242,"about_ca_system_score_codex":0.0011297765,"about_ca_system_score_gemma":0.0009920005,"threshold_uncertainty_score":0.01684469},"labels":[],"label_agreement":null},{"id":"W2134274165","doi":"10.1109/aqsdt.1992.205855","title":"SQEngineer: a methodology and tool for specifying and engineering software quality","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"St. Thomas University","funders":"","keywords":"Software engineering; Quality (philosophy); Software quality; Computer science; Software quality control; Software; Software quality analyst; Software requirements; Product (mathematics); Software construction; Software development; Programming language","score_opus":0.10247498171315238,"score_gpt":0.33139725545329135,"score_spread":0.22892227374013896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134274165","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00056679576,0.000084250365,0.98709315,0.00011575701,0.00002051705,0.00019921371,0.00023743439,0.010683622,0.0009992368],"genre_scores_gemma":[0.010739002,0.00023768756,0.98428947,0.00009181141,0.000024945088,0.000417124,0.001115756,0.0016435373,0.0014406831],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98079485,0.0063756225,0.0027337524,0.0013225427,0.008252438,0.0005208253],"domain_scores_gemma":[0.9763357,0.012167895,0.0025140587,0.0039149127,0.004651016,0.00041636743],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015543305,0.0022713004,0.001421464,0.006348702,0.0011134583,0.004619244,0.0036043786,0.0020291284,0.007943426],"category_scores_gemma":[0.03067936,0.0021881058,0.0022411647,0.0033790316,0.0026412797,0.0061963014,0.0032504052,0.0030459606,0.0041652215],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018569952,0.00029434531,0.003589817,0.0026142267,0.0002153536,0.00084137626,0.002774132,0.041727763,0.01615184,0.13555604,0.053565264,0.7424841],"study_design_scores_gemma":[0.00036481753,0.0006177904,0.0027962103,0.001562007,0.00020875239,0.002742931,0.00057776296,0.3402399,0.05752231,0.15492871,0.43798888,0.00044985925],"about_ca_topic_score_codex":0.0033535664,"about_ca_topic_score_gemma":0.0035975163,"teacher_disagreement_score":0.015543305,"about_ca_system_score_codex":0.0016071384,"about_ca_system_score_gemma":0.0052472875,"threshold_uncertainty_score":0.0822019},"labels":[],"label_agreement":null},{"id":"W2134502495","doi":"10.1109/ase.2009.23","title":"Using String Distances for Test Case Prioritisation","year":2009,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Test suite; Computer science; String (physics); Property (philosophy); Code (set theory); Test case; Test (biology); Random testing; Code coverage; Algorithm; Data mining; Theoretical computer science; Machine learning; Programming language; Software; Mathematics","score_opus":0.1170776203460329,"score_gpt":0.3653392665077053,"score_spread":0.2482616461616724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134502495","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053646743,0.00059169,0.939862,0.00023605894,0.00009161291,0.000269915,0.00027252454,0.0027501336,0.0022793298],"genre_scores_gemma":[0.33328107,0.0003017369,0.6635685,0.00012916542,0.00008037858,0.00028610532,0.00074198894,0.00043390298,0.0011771367],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9852055,0.0055596884,0.0013345364,0.0016386721,0.0058993516,0.00036237558],"domain_scores_gemma":[0.9657278,0.021590255,0.0037728073,0.003785797,0.0043910444,0.00073226134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060361135,0.0015549229,0.001238257,0.007125573,0.00070664764,0.0023553506,0.0021136247,0.0012872233,0.0027796782],"category_scores_gemma":[0.043642044,0.000376998,0.0007723096,0.005738849,0.0013538471,0.0035369042,0.0021716529,0.0013695301,0.0007883628],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012018296,0.0002724986,0.006014182,0.00052585476,0.00019540492,0.0003281988,0.0004146666,0.06989428,0.036092915,0.03228426,0.002501469,0.85027456],"study_design_scores_gemma":[0.00039540275,0.001818629,0.008203784,0.00017233028,0.0002381776,0.0015013088,0.00039048173,0.74075454,0.10984466,0.11770623,0.018732859,0.00024165577],"about_ca_topic_score_codex":0.0013181777,"about_ca_topic_score_gemma":0.0011360039,"teacher_disagreement_score":0.007125573,"about_ca_system_score_codex":0.0013122539,"about_ca_system_score_gemma":0.0013940386,"threshold_uncertainty_score":0.0319224},"labels":[],"label_agreement":null},{"id":"W2134872829","doi":"10.1109/csmr.2004.1281420","title":"Source code modularization using lattice of concept slices","year":2004,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Modular programming; Computer science; Program slicing; Programming language; Code refactoring; Software maintenance; Static program analysis; Slicing; Program analysis; Source code; Modular design; Lattice (music); Theoretical computer science; Software engineering; Software; Software system; Software development","score_opus":0.03697561773179269,"score_gpt":0.2800877208658379,"score_spread":0.2431121031340452,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134872829","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017555298,0.00007683548,0.980626,0.000081709695,0.0000115217545,0.000059162514,0.00006228138,0.00089217455,0.0006350199],"genre_scores_gemma":[0.13497336,0.0001274845,0.8634347,0.00003069108,0.000012680601,0.00012126084,0.00028568914,0.0001582033,0.00085589534],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.998582,0.0004353348,0.00014151554,0.00026709496,0.0004496025,0.00012443277],"domain_scores_gemma":[0.99730563,0.001166377,0.000336676,0.00061506254,0.00040998423,0.00016635309],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018444537,0.0004706052,0.00050711667,0.0015032628,0.0004418702,0.0013936722,0.00085672643,0.0003755673,0.001452986],"category_scores_gemma":[0.004808445,0.0004595665,0.0013533243,0.001245116,0.0015396453,0.0022467556,0.0018182799,0.00097140577,0.00036516663],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041299188,0.00014681942,0.0031291454,0.00038744972,0.00007678316,0.00048599124,0.0017969755,0.09345541,0.046574198,0.34247106,0.0033860311,0.50767714],"study_design_scores_gemma":[0.00009686923,0.00031128648,0.0010130686,0.00014757924,0.000077548975,0.0007795934,0.00048066588,0.58739156,0.052988477,0.32359865,0.033035472,0.00007929266],"about_ca_topic_score_codex":0.0015727937,"about_ca_topic_score_gemma":0.0011962297,"teacher_disagreement_score":0.0018444537,"about_ca_system_score_codex":0.0006168119,"about_ca_system_score_gemma":0.001183897,"threshold_uncertainty_score":0.009754479},"labels":[],"label_agreement":null},{"id":"W2135275739","doi":"10.1007/978-3-540-45221-8_22","title":"Towards Automated Support for Deriving Test Data from UML Statecharts","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Automation; Unified Modeling Language; Test case; Programming language; Normalization (sociology); Path (computing); Set (abstract data type); Event (particle physics); Data mining; Software engineering; Software; Machine learning","score_opus":0.04968096074950672,"score_gpt":0.30316102955412755,"score_spread":0.2534800688046208,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2135275739","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009578518,0.00008510385,0.9360114,0.00015821897,0.000029907358,0.00015610493,0.0008395717,0.052195,0.00094625825],"genre_scores_gemma":[0.14398196,0.0001640562,0.841667,0.00019465422,0.000053511336,0.00040746914,0.005823443,0.005498395,0.002209529],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9941446,0.0017115081,0.0007296622,0.00082391733,0.002238998,0.00035122404],"domain_scores_gemma":[0.9503201,0.03446885,0.0024576832,0.0075872107,0.004741727,0.0004243741],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046421746,0.0018803406,0.0016487223,0.004819785,0.0006945173,0.0035678619,0.004446965,0.0025354056,0.008338333],"category_scores_gemma":[0.037457474,0.0018860258,0.002289931,0.0026885709,0.0011602632,0.0049804663,0.0038895851,0.002851497,0.0044474737],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006632724,0.00061241374,0.0070506646,0.0010803427,0.00024111099,0.0009583787,0.0013034572,0.038296692,0.036973383,0.028383322,0.021588149,0.86284876],"study_design_scores_gemma":[0.00040125035,0.00021348952,0.0021893978,0.00033997407,0.00021498633,0.0008025615,0.0002880904,0.79051363,0.116447456,0.062514074,0.025919212,0.00015586491],"about_ca_topic_score_codex":0.004547173,"about_ca_topic_score_gemma":0.0064887796,"teacher_disagreement_score":0.008338333,"about_ca_system_score_codex":0.0010026523,"about_ca_system_score_gemma":0.0018861988,"threshold_uncertainty_score":0.027894497},"labels":[],"label_agreement":null},{"id":"W2135469438","doi":"10.1109/hase.2007.10","title":"Enhanced Traverse of Web Pages","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Traverse; Hyperlink; Computer science; Web testing; Web page; Reliability (semiconductor); Sequence (biology); Web site; Test (biology); World Wide Web; Web navigation; Information retrieval; The Internet; Data mining; Web development; Web application security","score_opus":0.02007141437038364,"score_gpt":0.2660981734370293,"score_spread":0.24602675906664564,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2135469438","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45938042,0.0004019331,0.51628613,0.0001157096,0.00005454269,0.00019534983,0.0004888044,0.016473824,0.0066033364],"genre_scores_gemma":[0.7312495,0.00016129634,0.26247606,0.000082195955,0.000014812911,0.00012577737,0.0011118827,0.0008028699,0.003975571],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988832,0.00033427306,0.00008575523,0.00024519744,0.00036635736,0.00008526682],"domain_scores_gemma":[0.99533087,0.0019462943,0.0003650956,0.0013817922,0.00083682145,0.00013911807],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006285084,0.00047214844,0.00045743713,0.0007780289,0.0002554286,0.0005190339,0.0008439694,0.0006481882,0.002682502],"category_scores_gemma":[0.005097172,0.00028922566,0.00047456016,0.00055905536,0.00026430763,0.0011870628,0.00066332263,0.00055359903,0.00087905704],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010873092,0.0006795077,0.009661631,0.00048393823,0.00007995199,0.0008774383,0.000383176,0.040332474,0.37465352,0.0045173136,0.0028636248,0.56438005],"study_design_scores_gemma":[0.00012475894,0.0013288077,0.012916714,0.00005879464,0.000095680356,0.002693321,0.0000936655,0.5168889,0.4458173,0.0065796194,0.013342262,0.00006023296],"about_ca_topic_score_codex":0.0009352905,"about_ca_topic_score_gemma":0.0011233157,"teacher_disagreement_score":0.002682502,"about_ca_system_score_codex":0.000268794,"about_ca_system_score_gemma":0.0004425396,"threshold_uncertainty_score":0.0089738965},"labels":[],"label_agreement":null},{"id":"W2136059906","doi":"10.1109/iceccs.2005.29","title":"Coping with Legacy System Migration Complexity","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Coping (psychology); Computer science; Legacy system; Psychology; Operating system; Software","score_opus":0.03662286275746241,"score_gpt":0.258066197491768,"score_spread":0.2214433347343056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136059906","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29891986,0.0018300227,0.673146,0.0073452448,0.00021905437,0.00035140957,0.00008216851,0.0033059497,0.014800293],"genre_scores_gemma":[0.6679514,0.0013553576,0.32070005,0.00070022425,0.00026872268,0.0002120612,0.00024991136,0.00052474544,0.008037507],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972242,0.0005509676,0.00019003706,0.00033717335,0.0013432925,0.00035430028],"domain_scores_gemma":[0.98843503,0.0040443474,0.0019979314,0.0026317495,0.0023224521,0.0005685343],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026215669,0.00074425974,0.0006655529,0.0020234385,0.0018909366,0.00293371,0.0022272703,0.0011216651,0.0012395191],"category_scores_gemma":[0.016416162,0.00051199255,0.0007185709,0.0017615922,0.00088907516,0.0048614712,0.0041919043,0.0023881998,0.00035140576],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025668062,0.00046393252,0.020246863,0.00075612974,0.00019462773,0.0021814462,0.0069108345,0.057765633,0.029675525,0.03392505,0.011036755,0.8365865],"study_design_scores_gemma":[0.00017585434,0.0006005216,0.029329224,0.0005007517,0.0004781511,0.007405011,0.0071870564,0.6537313,0.040860754,0.11876291,0.14071634,0.000252166],"about_ca_topic_score_codex":0.0019406852,"about_ca_topic_score_gemma":0.0028241875,"teacher_disagreement_score":0.00293371,"about_ca_system_score_codex":0.0010980138,"about_ca_system_score_gemma":0.0022584102,"threshold_uncertainty_score":0.013864279},"labels":[],"label_agreement":null},{"id":"W2136921053","doi":"10.1109/tse.2010.58","title":"The Effects of Time Constraints on Test Case Prioritization: A Series of Controlled Experiments","year":2010,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":166,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Regression testing; Prioritization; Computer science; Context (archaeology); Risk-based testing; Reliability engineering; Process (computing); Software; Regression analysis; Data mining; Risk analysis (engineering); Machine learning; Software development; Engineering; Management science; Software construction","score_opus":0.005484845015205005,"score_gpt":0.2193811498363502,"score_spread":0.21389630482114522,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136921053","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9740252,0.00038227,0.011603265,0.00020656876,0.0003282995,0.010084174,0.0005800266,0.00029924072,0.0024911305],"genre_scores_gemma":[0.872146,0.00060133537,0.065419614,0.0012052848,0.00033703307,0.05410937,0.00113904,0.0002570871,0.0047852406],"study_design_codex":"nonrandomized_trial","study_design_gemma":"nonrandomized_trial","domain_scores_codex":[0.9861057,0.0058070086,0.0018751144,0.0027173096,0.0022042894,0.0012906112],"domain_scores_gemma":[0.78933585,0.17330246,0.0181555,0.008236971,0.00777192,0.0031972928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014764209,0.0034786686,0.0017620694,0.0012345327,0.0013124457,0.0021484618,0.004320335,0.003055643,0.0058646575],"category_scores_gemma":[0.065497786,0.0013504918,0.0016918764,0.0013101477,0.0027290937,0.00241166,0.0017960834,0.0048815105,0.0006025335],"study_design_candidate":"nonrandomized_trial","study_design_consensus":"nonrandomized_trial","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.18169585,0.33123028,0.016931133,0.0049622147,0.0022530088,0.0011127874,0.008107449,0.045927897,0.2677897,0.006374319,0.0049967812,0.12861867],"study_design_scores_gemma":[0.06333205,0.6317912,0.046115436,0.0005238389,0.0030226633,0.0003337327,0.0019960832,0.055294324,0.16689493,0.013361185,0.016385296,0.0009493094],"about_ca_topic_score_codex":0.0024355275,"about_ca_topic_score_gemma":0.0028702493,"teacher_disagreement_score":0.014764209,"about_ca_system_score_codex":0.0025654656,"about_ca_system_score_gemma":0.002879321,"threshold_uncertainty_score":0.07808149},"labels":[],"label_agreement":null},{"id":"W2137519880","doi":"10.1109/ase.2002.1115018","title":"System testing for object-oriented frameworks using hook technology","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Test suite; Object-oriented programming; Software engineering; Generator (circuit theory); Suite; Hook; Software bug; Test case; Software; Distributed computing; Embedded system; Programming language; Engineering","score_opus":0.04378043057043085,"score_gpt":0.27130595706913574,"score_spread":0.2275255264987049,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2137519880","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23861538,0.00027083114,0.75083846,0.00017297985,0.000019698227,0.00015500358,0.0000685988,0.008035792,0.0018233726],"genre_scores_gemma":[0.8041875,0.00013846483,0.1942371,0.000049084596,0.000008895495,0.00012622621,0.0002314762,0.00025653816,0.0007647404],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9980563,0.0009236388,0.0001128358,0.0001389855,0.0006449951,0.00012340189],"domain_scores_gemma":[0.9931757,0.005005658,0.0006865215,0.0007282276,0.00030215,0.00010169355],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018363423,0.0005588857,0.00039831235,0.0011652224,0.00043418826,0.00080791925,0.0009368726,0.0008250606,0.0012331095],"category_scores_gemma":[0.009422474,0.0003560888,0.0007066688,0.00060073816,0.0009280245,0.0012773265,0.00076739036,0.0006079441,0.00023205981],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008492278,0.0007452788,0.024240367,0.0006075964,0.00019371588,0.002211576,0.0013845413,0.19081336,0.20463419,0.05323975,0.0032255931,0.5178548],"study_design_scores_gemma":[0.00018969103,0.0016341874,0.007269489,0.00012169599,0.00011518451,0.0014934463,0.00015321559,0.80206203,0.15247375,0.029690772,0.0047105798,0.00008599502],"about_ca_topic_score_codex":0.0010710377,"about_ca_topic_score_gemma":0.0010435679,"teacher_disagreement_score":0.0018363423,"about_ca_system_score_codex":0.0005342797,"about_ca_system_score_gemma":0.0005479071,"threshold_uncertainty_score":0.009711623},"labels":[],"label_agreement":null},{"id":"W2138203448","doi":"10.1109/icsecompanion.2007.56","title":"On Sufficiency of Mutants","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Mutation; Mutation testing; Computer science; Test suite; Set (abstract data type); Selection (genetic algorithm); Variable (mathematics); Process (computing); Regression analysis; Data mining; Algorithm; Artificial intelligence; Machine learning; Mathematics; Test case; Programming language; Genetics; Biology","score_opus":0.014024968195497286,"score_gpt":0.2740244165764556,"score_spread":0.2599994483809583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2138203448","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26123995,0.00047035233,0.7319868,0.00075999566,0.00005013843,0.0002081056,0.0003604762,0.0012409611,0.0036832062],"genre_scores_gemma":[0.8169769,0.00026181978,0.17996165,0.0003998257,0.000045067944,0.00035270365,0.0006927012,0.00043479796,0.0008745184],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9821488,0.008059346,0.0015114415,0.0026322228,0.004667642,0.000980493],"domain_scores_gemma":[0.79275715,0.16645098,0.0075071007,0.022765659,0.008774363,0.0017447774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01256977,0.0014623397,0.0018495049,0.0022823059,0.0010112036,0.0013722171,0.0017990634,0.0019283831,0.0021013261],"category_scores_gemma":[0.09976139,0.00078592723,0.0015080674,0.00097453076,0.0033784728,0.0039433967,0.002216242,0.0021989213,0.00035188414],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002619152,0.0010428078,0.035208218,0.0011412618,0.00030991202,0.001661822,0.0013239789,0.30595198,0.09805428,0.32583353,0.00542105,0.22143205],"study_design_scores_gemma":[0.00025267698,0.0012056705,0.0048109503,0.00024347517,0.00020318042,0.0019729964,0.00026386365,0.7102263,0.04573788,0.2297987,0.00520058,0.0000836322],"about_ca_topic_score_codex":0.000518343,"about_ca_topic_score_gemma":0.00056023983,"teacher_disagreement_score":0.01256977,"about_ca_system_score_codex":0.00086026435,"about_ca_system_score_gemma":0.0017338251,"threshold_uncertainty_score":0.06647611},"labels":[],"label_agreement":null},{"id":"W2138693312","doi":"10.1109/agile.2011.23","title":"Rule-Based Exploratory Testing of Graphical User Interfaces","year":2011,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Graphical user interface testing; Keyword-driven testing; Automation; Graphical user interface; Manual testing; Software engineering; Exploratory research; Test (biology); Software bug; Human–computer interaction; Data mining; User interface; Programming language; Software; User interface design; Software development; Engineering","score_opus":0.08777935105846328,"score_gpt":0.2585833439144655,"score_spread":0.17080399285600223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2138693312","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09701222,0.00033440746,0.8917521,0.00011524906,0.000031172298,0.00030276785,0.00012804517,0.007498084,0.0028258383],"genre_scores_gemma":[0.56806314,0.00020661112,0.4290866,0.000183756,0.000024236868,0.00023870022,0.00042218843,0.00052585994,0.0012488856],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99132556,0.0038500172,0.000604561,0.0009004767,0.0030136707,0.00030564805],"domain_scores_gemma":[0.9552307,0.03185434,0.0022542742,0.0059880647,0.004178864,0.00049380126],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045285844,0.0008562543,0.0007282138,0.0010630065,0.00028209,0.0013305658,0.0025524173,0.0009243142,0.0011099217],"category_scores_gemma":[0.031616613,0.0004027322,0.0007222232,0.00048384533,0.0008489063,0.0017960174,0.00087504904,0.0009967767,0.0005245811],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009322623,0.0008664831,0.01225701,0.0006712266,0.00019942607,0.0016221153,0.0019358195,0.060079306,0.17017058,0.011669872,0.0027388826,0.73685706],"study_design_scores_gemma":[0.00023358146,0.0019337705,0.009435719,0.0002857087,0.00015553708,0.0027322732,0.00028070546,0.73017067,0.21223922,0.030399831,0.011903832,0.00022911538],"about_ca_topic_score_codex":0.00081191043,"about_ca_topic_score_gemma":0.0009294283,"teacher_disagreement_score":0.0045285844,"about_ca_system_score_codex":0.00031498075,"about_ca_system_score_gemma":0.00051549164,"threshold_uncertainty_score":0.023949742},"labels":[],"label_agreement":null},{"id":"W2139217039","doi":"10.1145/379605.379630","title":"Evaluating explicitly context-sensitive program slicing","year":2001,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"McGill University","keywords":"Context (archaeology); Computer science; Sensitivity (control systems); Program slicing; Slicing; Programming language; World Wide Web; History; Engineering","score_opus":0.10189528572067504,"score_gpt":0.3893126797833766,"score_spread":0.28741739406270156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2139217039","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7426289,0.0012821829,0.24874036,0.00018806478,0.00006351447,0.0001352486,0.00022929332,0.0036397665,0.0030926566],"genre_scores_gemma":[0.90655214,0.00024265863,0.09218727,0.00005766023,0.000019643814,0.0000411671,0.0002731773,0.0002454708,0.00038079097],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9941203,0.002016174,0.0003057285,0.00045701684,0.0025437581,0.0005570037],"domain_scores_gemma":[0.97480166,0.015557198,0.0024171672,0.0035103932,0.0030790195,0.00063457375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00395757,0.0009970161,0.0009236112,0.001213592,0.00037064165,0.000977404,0.0013000427,0.001132944,0.0015477644],"category_scores_gemma":[0.028474612,0.0006227345,0.0005199153,0.0009535094,0.0016953633,0.0031186917,0.0013799143,0.0008202095,0.00017444318],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018711761,0.00032779534,0.028678836,0.0008304671,0.0003523897,0.0010462666,0.0006763549,0.62156934,0.13661376,0.031407144,0.00128842,0.17533809],"study_design_scores_gemma":[0.00008704456,0.0006821038,0.005680814,0.00007898967,0.0001295241,0.00025412676,0.00017493583,0.8900542,0.08289631,0.018532328,0.001369652,0.00005994787],"about_ca_topic_score_codex":0.0036873762,"about_ca_topic_score_gemma":0.005226476,"teacher_disagreement_score":0.00395757,"about_ca_system_score_codex":0.0011523586,"about_ca_system_score_gemma":0.0013202903,"threshold_uncertainty_score":0.020929873},"labels":[],"label_agreement":null},{"id":"W2139640569","doi":"10.1109/ccece.1998.682749","title":"Extended TTCN in software testing","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; White-box testing; Integration testing; Non-regression testing; Software reliability testing; Test strategy; System integration testing; Manual testing; Keyword-driven testing; Software performance testing; Test Management Approach; Software engineering; Black-box testing; Unit testing; Regression testing; Development testing; Conformance testing; Model-based testing; Process (computing); Software construction; Software; Test case; Programming language; Software development; Operating system","score_opus":0.05911903413093896,"score_gpt":0.2531458989008313,"score_spread":0.1940268647698923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2139640569","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005299824,0.0005965876,0.98266983,0.00036823642,0.00015673182,0.00014414887,0.00006850189,0.00085316563,0.009843008],"genre_scores_gemma":[0.22420715,0.0011107153,0.7623843,0.0010137802,0.00022706082,0.00072671135,0.0003939811,0.00059309,0.009343233],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98562545,0.0062091216,0.0011774198,0.0015352049,0.004933047,0.00051977823],"domain_scores_gemma":[0.9832388,0.009546241,0.0008826829,0.0031026911,0.00280007,0.00042952527],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0075297104,0.0009113225,0.0007139719,0.0021479616,0.0007283967,0.0020739143,0.0020220033,0.0013896835,0.0037775394],"category_scores_gemma":[0.017345876,0.00054741796,0.0015944213,0.0023508687,0.0040580267,0.005706573,0.0034148977,0.0033916202,0.0011246026],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020622845,0.000111882095,0.00096526294,0.0005130331,0.000034021192,0.0008122761,0.0008365553,0.052869756,0.0054288562,0.7519768,0.003506439,0.18273892],"study_design_scores_gemma":[0.00006654477,0.00020595656,0.0002936821,0.0003531621,0.00006357423,0.0010263361,0.000104047125,0.292349,0.0067662164,0.6185425,0.08015046,0.000078434394],"about_ca_topic_score_codex":0.0048747095,"about_ca_topic_score_gemma":0.0025954058,"teacher_disagreement_score":0.0075297104,"about_ca_system_score_codex":0.0023757492,"about_ca_system_score_gemma":0.0026743463,"threshold_uncertainty_score":0.039821386},"labels":[],"label_agreement":null},{"id":"W2140714756","doi":"10.1109/iceccs.2008.17","title":"On Extracting Tests from a Testable Model in the Context of Domain Engineering","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Domain (mathematical analysis); Domain engineering; Computer science; Domain analysis; Domain model; Executable; Model-based testing; Context (archaeology); Test case; Feature-oriented domain analysis; Data mining; Software engineering; Reliability engineering; Software; Software system; Machine learning; Programming language; Domain knowledge; Engineering; Mathematics; Component-based software engineering","score_opus":0.03282684462641823,"score_gpt":0.2453074403543252,"score_spread":0.21248059572790695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140714756","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0054503004,0.00008533964,0.99296415,0.0001788573,0.000008838779,0.000120872566,0.00008097197,0.0004334818,0.00067722367],"genre_scores_gemma":[0.09968627,0.00038108774,0.8976663,0.00013314543,0.000036778918,0.00036469084,0.0008322793,0.00023872196,0.0006606932],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99573696,0.0015131012,0.00028401794,0.000388116,0.0019147651,0.00016298988],"domain_scores_gemma":[0.970697,0.022547273,0.0015052696,0.0036423877,0.0014240383,0.00018411403],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028809453,0.0015057607,0.0010428369,0.0026150518,0.00071267964,0.001826907,0.0017505133,0.0019765378,0.0017706747],"category_scores_gemma":[0.028878557,0.0007012252,0.0016203598,0.0024758915,0.0025648444,0.0033384108,0.0015957473,0.0024027159,0.00052877696],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024133048,0.00052858266,0.0054244217,0.001314039,0.00012480721,0.00314146,0.0016475448,0.2911208,0.035799224,0.23087941,0.0030722665,0.426706],"study_design_scores_gemma":[0.00005335809,0.00024098373,0.0010908156,0.00038344998,0.00007916648,0.0011563777,0.00028901806,0.75549704,0.03878278,0.18983068,0.012523164,0.00007314333],"about_ca_topic_score_codex":0.0017798701,"about_ca_topic_score_gemma":0.0021729667,"teacher_disagreement_score":0.0028809453,"about_ca_system_score_codex":0.00096508814,"about_ca_system_score_gemma":0.0014215539,"threshold_uncertainty_score":0.01523608},"labels":[],"label_agreement":null},{"id":"W2141291123","doi":"10.1145/1108768.1108795","title":"An empirical framework for comparing effectiveness of testing and property-based formal analysis","year":2005,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Debugging; Computer science; Software engineering; Software bug; Formal methods; Software testing; Empirical research; Formal specification; Software; Reliability engineering; Programming language; Engineering; Mathematics","score_opus":0.042738907365135706,"score_gpt":0.30932317573452706,"score_spread":0.26658426836939136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141291123","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33028036,0.0052247094,0.6128856,0.002738899,0.00039716283,0.006173338,0.0025053706,0.0005606965,0.039233807],"genre_scores_gemma":[0.8599052,0.0006723963,0.12898977,0.000597795,0.00023125642,0.007729822,0.0010078738,0.00018160943,0.00068433135],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.6888434,0.24602483,0.014069057,0.006463504,0.042543065,0.0020561188],"domain_scores_gemma":[0.16678585,0.76918113,0.020635977,0.02476238,0.017215839,0.0014188838],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.17228676,0.0013755817,0.0014125339,0.010037864,0.0010970064,0.0044108126,0.0022886272,0.0027076555,0.0027246561],"category_scores_gemma":[0.5867899,0.00046707081,0.0017898413,0.007516942,0.007457251,0.0084153805,0.0037149484,0.0031144074,0.0005881826],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006552163,0.005702456,0.31239763,0.0041366266,0.0041752937,0.00042433303,0.007571692,0.051688265,0.0067258426,0.30952296,0.0061783036,0.28492445],"study_design_scores_gemma":[0.002851192,0.032364763,0.29730257,0.0030366934,0.0019922967,0.0018400376,0.007322393,0.29091468,0.012438836,0.31539723,0.033901174,0.00063816726],"about_ca_topic_score_codex":0.0009997279,"about_ca_topic_score_gemma":0.0005247871,"teacher_disagreement_score":0.82771325,"about_ca_system_score_codex":0.0022544381,"about_ca_system_score_gemma":0.00198233,"threshold_uncertainty_score":0.9111504},"labels":[],"label_agreement":null},{"id":"W2141469470","doi":"10.1109/apaq.2000.883775","title":"Control of nondeterminism in testing distributed multithreaded programs","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Java; Distributed computing; Programming language; Common Object Request Broker Architecture; Test case; Set (abstract data type); System under test; Constraint (computer-aided design); Control (management)","score_opus":0.06328583041465877,"score_gpt":0.2540138546615939,"score_spread":0.1907280242469351,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141469470","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13588591,0.00020318577,0.85890985,0.000272097,0.000021972523,0.00018593844,0.000035597743,0.0025006416,0.0019847557],"genre_scores_gemma":[0.8659339,0.00006961265,0.1329297,0.00007497169,0.000015006971,0.0001805415,0.000040472863,0.00025771116,0.00049814576],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9798236,0.01068592,0.0012846728,0.0023908804,0.004731434,0.0010834961],"domain_scores_gemma":[0.8895436,0.090578236,0.0050278,0.010875937,0.0031223292,0.0008520967],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013037738,0.00079000223,0.0008951055,0.0010075888,0.00097620056,0.0028217204,0.002357585,0.0010154839,0.00091534335],"category_scores_gemma":[0.06170131,0.0007816811,0.00076925335,0.000654774,0.00620382,0.004837649,0.0021290062,0.0020853917,0.00011635134],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020911626,0.00068149454,0.025102934,0.000992788,0.000228877,0.0011010471,0.003956045,0.36574534,0.08999511,0.23651423,0.0011814906,0.27240944],"study_design_scores_gemma":[0.0002211912,0.00038243664,0.0017729893,0.000089647096,0.000086681874,0.00028321997,0.00015595315,0.82632804,0.08628619,0.0818249,0.0024815004,0.00008720606],"about_ca_topic_score_codex":0.004693088,"about_ca_topic_score_gemma":0.0038389717,"teacher_disagreement_score":0.013037738,"about_ca_system_score_codex":0.0017722246,"about_ca_system_score_gemma":0.0025457635,"threshold_uncertainty_score":0.06895101},"labels":[],"label_agreement":null},{"id":"W2142266413","doi":"10.5555/998675.999415","title":"Using simulation to empirically investigate test coverage criteria based on statechart","year":2004,"lang":"en","type":"article","venue":"International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":81,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Code coverage; Class (philosophy); Test (biology); Test case; Reliability engineering; Fault detection and isolation; Test strategy; Data mining; Machine learning; Artificial intelligence; Programming language; Software; Engineering","score_opus":0.08716524489143154,"score_gpt":0.3492094312045713,"score_spread":0.26204418631313975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142266413","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.920634,0.00035063273,0.072308004,0.0002797819,0.000015527903,0.0003514563,0.0005394383,0.00017346872,0.0053477935],"genre_scores_gemma":[0.9843237,0.00008664277,0.014997738,0.000020257065,0.000003455101,0.00020762511,0.00022960732,0.000012757452,0.0001183195],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98648155,0.009789949,0.00072149193,0.000617566,0.0018739329,0.0005155331],"domain_scores_gemma":[0.6261046,0.35339603,0.007147089,0.0067996,0.006010451,0.00054225745],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016247153,0.0009200234,0.0007054651,0.0032023173,0.00041373377,0.0015214694,0.0010619304,0.0013807926,0.0017199886],"category_scores_gemma":[0.12305667,0.00046472135,0.0009635579,0.0027787823,0.0010192047,0.002487482,0.0009162495,0.0010407661,0.00014142362],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008560826,0.0006528771,0.05660278,0.00027293267,0.00038446716,0.00011446854,0.00043043747,0.8989832,0.0019871031,0.01774184,0.00040172122,0.021572176],"study_design_scores_gemma":[0.00010062584,0.0012430891,0.010723815,0.000067658024,0.00013993743,0.00008806279,0.00026549786,0.9747321,0.003218948,0.008781188,0.00060177356,0.00003743161],"about_ca_topic_score_codex":0.004847192,"about_ca_topic_score_gemma":0.005274597,"teacher_disagreement_score":0.016247153,"about_ca_system_score_codex":0.002767218,"about_ca_system_score_gemma":0.0013434201,"threshold_uncertainty_score":0.08592421},"labels":[],"label_agreement":null},{"id":"W2143742229","doi":"10.5555/2486788.2486974","title":"On extracting unit tests from interactive live programming sessions","year":2013,"lang":"en","type":"article","venue":"International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Leverage (statistics); Session (web analytics); Unit testing; Cluster analysis; Software testing; Software; Software engineering; Programming language; Machine learning; World Wide Web","score_opus":0.03945455755763936,"score_gpt":0.2990520090276418,"score_spread":0.2595974514700024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2143742229","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09559848,0.00035368634,0.893101,0.00017757942,0.000048561124,0.00038720216,0.0012769346,0.0065239407,0.002532695],"genre_scores_gemma":[0.32702726,0.00044371578,0.66186744,0.000099511104,0.00006696743,0.00053439406,0.0049361247,0.0013080531,0.0037165547],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9979463,0.00056560145,0.00016328298,0.00044140551,0.0007443517,0.00013895311],"domain_scores_gemma":[0.9776534,0.012701484,0.0021324493,0.003965234,0.0030899153,0.00045751448],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012686761,0.0009821109,0.0007013171,0.0036867948,0.0005945374,0.0015579069,0.0017506026,0.0010322655,0.0026725333],"category_scores_gemma":[0.020262608,0.00041151547,0.00046555442,0.0032276239,0.00081970904,0.0022158993,0.0012623313,0.0011600207,0.0015677565],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038355222,0.0002796125,0.014815529,0.00061251357,0.000066800254,0.001139563,0.0034304743,0.011599385,0.071874015,0.004983168,0.0032659757,0.88754946],"study_design_scores_gemma":[0.00013711287,0.0012779016,0.11042036,0.00077060267,0.0002110168,0.006293616,0.0051675164,0.4533481,0.30151442,0.05263734,0.06774037,0.00048170544],"about_ca_topic_score_codex":0.0016945177,"about_ca_topic_score_gemma":0.0033468348,"teacher_disagreement_score":0.0036867948,"about_ca_system_score_codex":0.0003753094,"about_ca_system_score_gemma":0.00058486644,"threshold_uncertainty_score":0.008940458},"labels":[],"label_agreement":null},{"id":"W2143791583","doi":"10.1109/apsec.2002.1183004","title":"Quality driven transformation compositions for object oriented migration","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Business process reengineering; Legacy system; Transformation (genetics); Model transformation; Object-oriented programming; Source code; Software engineering; Process (computing); Programming language; Object (grammar); Program transformation; Quality (philosophy); Set (abstract data type); Software system; Software quality; Software development; Software; Artificial intelligence; Engineering","score_opus":0.0352107065189372,"score_gpt":0.3172132540120987,"score_spread":0.2820025474931615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2143791583","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0071410174,0.0000899664,0.9890175,0.00009350326,0.000020188107,0.00010308338,0.000011986613,0.0009964026,0.0025262742],"genre_scores_gemma":[0.16337956,0.00022170652,0.8305485,0.00009791432,0.000023155419,0.0001858915,0.0001169986,0.00039425245,0.005031932],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99755365,0.000604117,0.00013900973,0.00022813004,0.001343398,0.00013164591],"domain_scores_gemma":[0.9981669,0.0005985432,0.00030728432,0.00042997792,0.0004207323,0.0000765873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030125508,0.00055417663,0.00036904286,0.0011299815,0.0008005656,0.0012176759,0.0008008761,0.0008712909,0.0026982478],"category_scores_gemma":[0.0059109735,0.00045116676,0.00090198073,0.00074615655,0.0011733163,0.0012308295,0.0015699569,0.0012461268,0.0011158054],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027558437,0.00029073344,0.0026754355,0.00038195297,0.000105999025,0.00091175974,0.0013355615,0.15177254,0.05777785,0.39779112,0.004021039,0.38266042],"study_design_scores_gemma":[0.00015473191,0.00023893554,0.0010100243,0.000109866174,0.00012346802,0.00079132913,0.0002519257,0.6935117,0.06846601,0.18288386,0.05239668,0.00006151683],"about_ca_topic_score_codex":0.0017229476,"about_ca_topic_score_gemma":0.0019652534,"teacher_disagreement_score":0.0030125508,"about_ca_system_score_codex":0.00080413226,"about_ca_system_score_gemma":0.0013662219,"threshold_uncertainty_score":0.015932083},"labels":[],"label_agreement":null},{"id":"W2144667363","doi":"10.1145/1145735.1145741","title":"Tool support for randomized unit testing","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Java; Unit testing; Class (philosophy); Unit (ring theory); Programming language; Software; Mathematics; Artificial intelligence","score_opus":0.033379706469621157,"score_gpt":0.2757351743738155,"score_spread":0.24235546790419435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2144667363","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040112976,0.00010998972,0.9332768,0.00018067296,0.000040988914,0.00014066235,0.0002023669,0.06049171,0.0015455075],"genre_scores_gemma":[0.12999928,0.0002261559,0.8569401,0.0004234715,0.00007914673,0.0007212322,0.001196097,0.008577376,0.0018370386],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9849225,0.0064736567,0.0023573413,0.0018826224,0.0035381946,0.0008255443],"domain_scores_gemma":[0.91184795,0.059401806,0.0036786303,0.019785007,0.0044162697,0.0008703514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016036263,0.0012917713,0.0016339063,0.002316953,0.000633209,0.0027683012,0.006118514,0.0017695045,0.009833542],"category_scores_gemma":[0.06489823,0.0016606054,0.0022314908,0.0015012022,0.001529318,0.0068780226,0.0039566155,0.0031230843,0.0031583088],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002051999,0.0011860739,0.009104128,0.002279327,0.0004139759,0.001685452,0.0012669946,0.066421114,0.042324726,0.16920766,0.043714643,0.66034395],"study_design_scores_gemma":[0.0010177636,0.00065891456,0.0019790174,0.00072771154,0.00018965686,0.0023194111,0.00015163071,0.68405783,0.06163969,0.15298024,0.09396138,0.00031680046],"about_ca_topic_score_codex":0.00083722,"about_ca_topic_score_gemma":0.0010663647,"teacher_disagreement_score":0.016036263,"about_ca_system_score_codex":0.0008040959,"about_ca_system_score_gemma":0.0018837138,"threshold_uncertainty_score":0.08480883},"labels":[],"label_agreement":null},{"id":"W2144931777","doi":"10.1109/hcc.2002.1046355","title":"A data-flow testing methodology for a dataflow based visual programming language","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Dataflow; Computer science; Programming language; Visual programming language; Visual language; Data flow diagram; High-level programming language; Programming paradigm; Database","score_opus":0.25640698913435905,"score_gpt":0.4144659342470359,"score_spread":0.15805894511267687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2144931777","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030183555,0.000024098623,0.99444157,0.000038451908,0.000009717846,0.00007312439,0.000020478681,0.0020750372,0.000299207],"genre_scores_gemma":[0.11246388,0.00007082294,0.88519996,0.000114820454,0.000015572772,0.00023772592,0.000152019,0.0005141152,0.0012310505],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974847,0.0006587675,0.00021872169,0.00041370277,0.0010940252,0.00013017903],"domain_scores_gemma":[0.99340665,0.0032948751,0.00066346605,0.001089736,0.0013514296,0.000193871],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028596665,0.0008568533,0.0005467445,0.0014258226,0.00052293413,0.0010994163,0.0024037082,0.00096258573,0.0025802548],"category_scores_gemma":[0.009280034,0.00047489977,0.00080526405,0.0005479724,0.0012436528,0.002209776,0.0012066294,0.0013583332,0.00046824192],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003213082,0.00060524297,0.0070091104,0.00085949636,0.00012016421,0.00083389255,0.0011546523,0.030463431,0.13881993,0.07787559,0.004681849,0.73725545],"study_design_scores_gemma":[0.00018082625,0.001120218,0.0029994715,0.0004084842,0.00015641119,0.0031997152,0.00022438912,0.60536534,0.28258315,0.06776885,0.035816874,0.00017628669],"about_ca_topic_score_codex":0.0012567976,"about_ca_topic_score_gemma":0.00108412,"teacher_disagreement_score":0.0028596665,"about_ca_system_score_codex":0.0005572405,"about_ca_system_score_gemma":0.0010701112,"threshold_uncertainty_score":0.015123546},"labels":[],"label_agreement":null},{"id":"W2145149095","doi":"10.1002/stvr.461","title":"Regression test suite prioritization using system models","year":2011,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Regression testing; Test suite; Computer science; Prioritization; Reliability engineering; Test (biology); Test case; Fault detection and isolation; System under test; Test Management Approach; Empirical research; Overhead (engineering); Suite; Regression analysis; Data mining; Machine learning; Artificial intelligence; Engineering; Software system; Statistics; Software; Programming language","score_opus":0.08413484429999112,"score_gpt":0.2648815431189903,"score_spread":0.18074669881899919,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145149095","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19735795,0.00039249365,0.7927605,0.00031141329,0.000032935473,0.00027071693,0.00019602191,0.0041514584,0.004526441],"genre_scores_gemma":[0.83300287,0.00011645614,0.16547835,0.000041131912,0.000016906086,0.00012778513,0.00025626228,0.00022134294,0.0007389513],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9939166,0.0034334392,0.00023685349,0.00042312787,0.0017089498,0.00028115758],"domain_scores_gemma":[0.9773501,0.016708946,0.0016827426,0.0019348817,0.0020657266,0.00025752344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005257537,0.0010585084,0.0007414755,0.0021619361,0.00027091222,0.0010771577,0.0011154748,0.0004625328,0.001777706],"category_scores_gemma":[0.021949288,0.0004777441,0.0007222638,0.001081807,0.00039705503,0.0013267562,0.0008196723,0.00097353256,0.0002456773],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004814765,0.0002798837,0.009404477,0.00015791623,0.00011559735,0.00011606793,0.00013761055,0.7876969,0.009746252,0.009891013,0.0015090597,0.18046382],"study_design_scores_gemma":[0.000025894271,0.00009559602,0.0006630533,0.000010279412,0.00002499352,0.000037146205,0.000011558931,0.99200165,0.0040802574,0.0025402976,0.00049831445,0.000010843089],"about_ca_topic_score_codex":0.0051153935,"about_ca_topic_score_gemma":0.0051773717,"teacher_disagreement_score":0.005257537,"about_ca_system_score_codex":0.0014784358,"about_ca_system_score_gemma":0.0014399134,"threshold_uncertainty_score":0.027804792},"labels":[],"label_agreement":null},{"id":"W2145733341","doi":"10.1145/2786805.2786858","title":"Assertions are strongly correlated with test suite effectiveness","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":101,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Test suite; Assertion; Test (biology); Computer science; Suite; Java; Test case; Code coverage; Code (set theory); Unit testing; Programming language; Algorithm; Machine learning; Software; Regression analysis","score_opus":0.029829427367251995,"score_gpt":0.2597681029273936,"score_spread":0.2299386755601416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145733341","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9846379,0.0009310448,0.010912118,0.00029089893,0.000021334532,0.00004812701,0.00030929816,0.0004895542,0.0023596985],"genre_scores_gemma":[0.9949485,0.00014602856,0.004023796,0.000034542023,0.00002790815,0.00003070834,0.0005308881,0.000084891115,0.00017277816],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.96731275,0.009785358,0.004322873,0.0027640802,0.01450468,0.0013102656],"domain_scores_gemma":[0.34112358,0.5635101,0.053689364,0.013332547,0.024086969,0.004257432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018886153,0.00084021257,0.00078174105,0.0049859304,0.00033548218,0.0017663639,0.0009264762,0.0010709261,0.0014747882],"category_scores_gemma":[0.2660828,0.00048061393,0.00081784197,0.00261146,0.001097516,0.0032935913,0.001256342,0.0012142378,0.00046743697],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060116686,0.0003185691,0.903013,0.0003789793,0.00059882866,0.00052039884,0.0007475669,0.01785992,0.008079708,0.0010627096,0.00080109003,0.06601793],"study_design_scores_gemma":[0.000054056487,0.0012108276,0.89856696,0.00014176888,0.00058707007,0.0016674335,0.00069289736,0.079654746,0.011700675,0.0037087682,0.0019339708,0.00008071618],"about_ca_topic_score_codex":0.0010998962,"about_ca_topic_score_gemma":0.0012082256,"teacher_disagreement_score":0.018886153,"about_ca_system_score_codex":0.0006190499,"about_ca_system_score_gemma":0.00070830423,"threshold_uncertainty_score":0.099880695},"labels":[],"label_agreement":null},{"id":"W2148177313","doi":"10.1109/iscas.2000.857100","title":"A methodology for validating digital circuits with mutation testing","year":2000,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Verilog; Mutation; Mutation testing; Functional testing; Hardware description language; VHDL; Task (project management); Digital electronics; Theoretical computer science; Computer engineering; Programming language; Reliability engineering; Electronic circuit; Embedded system; Field-programmable gate array; Engineering","score_opus":0.1306905846849651,"score_gpt":0.31829560941346813,"score_spread":0.18760502472850302,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2148177313","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028894516,0.00007609758,0.99577254,0.000039934366,0.000021193131,0.00007638405,0.000017768929,0.00059260335,0.0005140326],"genre_scores_gemma":[0.10266084,0.00026350495,0.8950134,0.00010673584,0.000027570044,0.0002928697,0.00013314474,0.00022248033,0.0012794695],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99553347,0.0013597059,0.00030517837,0.00044320102,0.0022002622,0.00015807552],"domain_scores_gemma":[0.99557894,0.0021777751,0.00043954444,0.001000875,0.0007345058,0.000068243884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024656176,0.0011172348,0.00066085736,0.0022613755,0.0006820971,0.0007693483,0.0014742897,0.0009470319,0.001198311],"category_scores_gemma":[0.00818894,0.00049319863,0.0010503097,0.00087616034,0.0021148892,0.0015624319,0.0010611381,0.0014006636,0.00045719036],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012275996,0.00019478158,0.0031310252,0.0005459139,0.00016968037,0.0006994861,0.00049807335,0.074331075,0.14705288,0.21225685,0.0024021196,0.5585954],"study_design_scores_gemma":[0.0001481268,0.0011954904,0.0024330448,0.00038673135,0.00022571739,0.004485336,0.00011845862,0.455048,0.30495736,0.161183,0.069602214,0.00021649638],"about_ca_topic_score_codex":0.0008343259,"about_ca_topic_score_gemma":0.0006740996,"teacher_disagreement_score":0.0024656176,"about_ca_system_score_codex":0.00062989246,"about_ca_system_score_gemma":0.0013060955,"threshold_uncertainty_score":0.013039589},"labels":[],"label_agreement":null},{"id":"W2149695823","doi":"10.1007/s10664-006-9026-0","title":"Empirical evaluation of optimization algorithms when used in goal-oriented automated test data generation techniques","year":2006,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Test Management Approach; Computer science; Keyword-driven testing; Test data; Software development; Test harness; Software; Software reliability testing; Automation; Test strategy; Non-regression testing; Data mining; Software construction; Software engineering; Engineering; Programming language","score_opus":0.06930117347575526,"score_gpt":0.3370234376690021,"score_spread":0.26772226419324685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149695823","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8740962,0.0012424221,0.1191417,0.00038127278,0.00004463692,0.0002726076,0.00032874304,0.0015346513,0.002957827],"genre_scores_gemma":[0.9209353,0.00019097347,0.07771669,0.00004668158,0.00001761189,0.00015147091,0.00042716527,0.00020017455,0.0003139954],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9747628,0.017536275,0.00157797,0.0014824348,0.004127826,0.00051268004],"domain_scores_gemma":[0.64706683,0.31571245,0.008952906,0.015896201,0.011543443,0.0008281814],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.017505858,0.0010282947,0.00064870145,0.0020068763,0.00045565434,0.0011852556,0.0014868038,0.0015983437,0.0009996707],"category_scores_gemma":[0.21532741,0.0004180518,0.00048345237,0.0019439888,0.0010329772,0.0024737183,0.0010317129,0.0013593483,0.0002745125],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039592637,0.0042913877,0.0783202,0.00095799385,0.0004498348,0.0001402851,0.0010062009,0.3768694,0.009446044,0.006148707,0.002845467,0.51556516],"study_design_scores_gemma":[0.00038087816,0.0018324736,0.022168344,0.00009159689,0.00020512886,0.00018972368,0.00018872917,0.957678,0.013418271,0.0026588137,0.0011486218,0.000039478076],"about_ca_topic_score_codex":0.0018611407,"about_ca_topic_score_gemma":0.002181814,"teacher_disagreement_score":0.9824941,"about_ca_system_score_codex":0.001011379,"about_ca_system_score_gemma":0.001143565,"threshold_uncertainty_score":0.092580914},"labels":[],"label_agreement":null},{"id":"W2151635819","doi":"10.1145/1276958.1277172","title":"Automatic mutation test input data generation via ant colony","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":109,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Ant colony optimization algorithms; Mutation; Context (archaeology); Genetic algorithm; Java; Code coverage; Test data; Data mining; Machine learning; Artificial intelligence; Software","score_opus":0.059541612366656615,"score_gpt":0.3142915271483908,"score_spread":0.2547499147817342,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2151635819","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12039215,0.00015885559,0.872851,0.00018688136,0.000035171845,0.00017226711,0.00003829595,0.004112982,0.00205245],"genre_scores_gemma":[0.69018906,0.00008044231,0.3081494,0.00006967866,0.000012250083,0.00018134544,0.00009799333,0.00014571613,0.0010741401],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99901104,0.00028850933,0.000057615256,0.00012547406,0.00043752108,0.00007982382],"domain_scores_gemma":[0.9979468,0.0010610137,0.0002178367,0.0003653217,0.00033951856,0.000069529575],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00080706936,0.00059653656,0.00073088426,0.00073186593,0.00028915616,0.00047289522,0.0012256643,0.00068795605,0.0010194052],"category_scores_gemma":[0.0036741288,0.00026160802,0.00036430152,0.00046940142,0.0004163484,0.000625209,0.0007813065,0.0006384706,0.00020060324],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021580569,0.00030931877,0.0035202615,0.00011877053,0.00007348556,0.00039124198,0.00018864106,0.4395267,0.097070806,0.0072631813,0.0017963882,0.44952542],"study_design_scores_gemma":[0.00002852139,0.00005150843,0.00038131492,0.0000029952953,0.000009266332,0.00008736338,0.000008632305,0.98283976,0.01473152,0.0010624405,0.0007880797,0.000008621438],"about_ca_topic_score_codex":0.0017650014,"about_ca_topic_score_gemma":0.0012156032,"teacher_disagreement_score":0.0017650014,"about_ca_system_score_codex":0.00041226446,"about_ca_system_score_gemma":0.00048229223,"threshold_uncertainty_score":0.0042682886},"labels":[],"label_agreement":null},{"id":"W2152665443","doi":"10.1109/icsm.1990.131335","title":"Non-intrusive tool for software testing and debugging: an actual industrial application example","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nortel (Canada)","funders":"","keywords":"Debugging; Computer science; Software engineering; Software; Programming language; Embedded system","score_opus":0.09142306802878755,"score_gpt":0.2741139128491273,"score_spread":0.18269084482033973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2152665443","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15448444,0.00060267403,0.8027062,0.0007890175,0.00012852994,0.0004465766,0.0001798483,0.018991048,0.02167171],"genre_scores_gemma":[0.48538002,0.00031586105,0.50049007,0.0002331063,0.00004091032,0.00017084401,0.00028761767,0.0010202427,0.012061366],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99832326,0.00049928034,0.000081997736,0.00013178945,0.0008354824,0.00012815397],"domain_scores_gemma":[0.99590486,0.0021051464,0.00013375543,0.00089209893,0.000777432,0.00018668156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013490745,0.00058475777,0.0004072457,0.0009652738,0.0006420155,0.00095084595,0.0013925768,0.0011657036,0.005191909],"category_scores_gemma":[0.0049948706,0.0003380237,0.0002906786,0.0012083311,0.0005728334,0.0011704988,0.00076486025,0.001020045,0.0012809174],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007593293,0.0013357353,0.0075978893,0.00054813636,0.000054450877,0.0055410676,0.0028582823,0.021349613,0.08516462,0.03594759,0.019850805,0.81899256],"study_design_scores_gemma":[0.0006486402,0.0039484985,0.010546817,0.00024364452,0.00018259864,0.013520518,0.0014876762,0.4021583,0.26123422,0.023079408,0.2827165,0.00023315642],"about_ca_topic_score_codex":0.0011459073,"about_ca_topic_score_gemma":0.001965691,"teacher_disagreement_score":0.005191909,"about_ca_system_score_codex":0.00036087222,"about_ca_system_score_gemma":0.0005010209,"threshold_uncertainty_score":0.017368734},"labels":[],"label_agreement":null},{"id":"W2153135840","doi":"10.1109/esem.2007.22","title":"Assessing, Comparing, and Combining Statechart- based testing and Structural testing: An Experiment","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Code coverage; Software testing; Test strategy; Code (set theory); Fault detection and isolation; Fault coverage; Source code; Reliability engineering; Class (philosophy); Test case; Software; Programming language; Machine learning; Artificial intelligence; Set (abstract data type); Engineering","score_opus":0.10540085681696369,"score_gpt":0.35147260285424115,"score_spread":0.24607174603727744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2153135840","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99066097,0.00004799677,0.007408828,0.000065332104,0.000025323367,0.0008421274,0.00015421346,0.00013704783,0.0006582419],"genre_scores_gemma":[0.96399254,0.00009694076,0.031900603,0.0001324125,0.00003803949,0.0021414044,0.0003573496,0.00006421279,0.0012765327],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9933802,0.003406393,0.00069024635,0.00091359805,0.0012831329,0.00032643136],"domain_scores_gemma":[0.9108684,0.07797879,0.002312065,0.0048247925,0.0027030725,0.0013127859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008714617,0.0011507105,0.0010306657,0.0006887459,0.00045712286,0.0009884194,0.0020391392,0.00218768,0.0026405023],"category_scores_gemma":[0.02975025,0.00073449203,0.0008176811,0.0006374543,0.0009248115,0.0022894377,0.0011870159,0.0012400114,0.000413891],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.07885104,0.13302915,0.03473119,0.002203408,0.0008957648,0.0008814061,0.007209324,0.09152328,0.41062358,0.005086683,0.0018647099,0.23310041],"study_design_scores_gemma":[0.020624436,0.4015029,0.03658406,0.00015943142,0.0012373864,0.00060799747,0.0015850611,0.18282801,0.3389198,0.005780858,0.009641953,0.00052809954],"about_ca_topic_score_codex":0.0011820197,"about_ca_topic_score_gemma":0.0010753978,"teacher_disagreement_score":0.008714617,"about_ca_system_score_codex":0.00072698825,"about_ca_system_score_gemma":0.0009836475,"threshold_uncertainty_score":0.04608786},"labels":[],"label_agreement":null},{"id":"W2153934195","doi":"10.1145/1068009.1068185","title":"Improving network applications security","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Buffer overflow; Computer science; Program slicing; Dependency graph; Static analysis; Dependency (UML); Call graph; Slicing; Static program analysis; Control flow graph; Exploit; Distributed computing; Data-flow analysis; Software; Graph; Computer security; Data flow diagram; Theoretical computer science; Software engineering; Programming language; Software development; Database","score_opus":0.009514283014754729,"score_gpt":0.24275605342952686,"score_spread":0.23324177041477212,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2153934195","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6080928,0.004649004,0.29732823,0.004676106,0.00019037879,0.00021561327,0.00025259555,0.0060193916,0.078575924],"genre_scores_gemma":[0.9513709,0.0011233109,0.04066781,0.00022586467,0.00004578844,0.000033772198,0.0001403326,0.00022464321,0.006167558],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990615,0.00025466442,0.000035353598,0.000105928535,0.00034369717,0.00019885595],"domain_scores_gemma":[0.99668616,0.0012273116,0.00036266082,0.0006404647,0.0009201757,0.00016317761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012721198,0.00076412485,0.00040927378,0.0011532156,0.00072167214,0.0016170108,0.0005815396,0.000615344,0.0051782588],"category_scores_gemma":[0.0072038514,0.00017707322,0.00032111586,0.00069752306,0.00042937996,0.0030841143,0.0014192524,0.0007974398,0.00093280745],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035323313,0.00038267547,0.017237436,0.00030013578,0.00009032188,0.00041090848,0.00054173754,0.11101891,0.09880672,0.036551666,0.008971485,0.72533476],"study_design_scores_gemma":[0.000103675135,0.00092837465,0.020688878,0.00026760352,0.0003358809,0.0014156016,0.0008194542,0.71655726,0.12844063,0.053663548,0.076693326,0.00008582627],"about_ca_topic_score_codex":0.0021029296,"about_ca_topic_score_gemma":0.0027228196,"teacher_disagreement_score":0.0051782588,"about_ca_system_score_codex":0.00078418566,"about_ca_system_score_gemma":0.0009155941,"threshold_uncertainty_score":0.017323077},"labels":[],"label_agreement":null},{"id":"W2155119310","doi":"10.1109/tim.2006.887405","title":"DOS Middleware Instrumentation for Ensuring Reproducibility of Testing Procedures","year":2007,"lang":"en","type":"article","venue":"IEEE Transactions on Instrumentation and Measurement","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Middleware (distributed applications); Distributed computing; Layer (electronics); Operating system; Embedded system","score_opus":0.09772365009063337,"score_gpt":0.3037248750879295,"score_spread":0.20600122499729612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155119310","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037053656,0.00024033333,0.9389115,0.00049330294,0.0001651793,0.00045724856,0.00007962056,0.017262084,0.0053371433],"genre_scores_gemma":[0.5452004,0.00020203477,0.44815782,0.00048772732,0.00012879299,0.0007911437,0.00024019695,0.0019089752,0.002882883],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9740703,0.008731197,0.0026722683,0.0031685254,0.010083641,0.0012741216],"domain_scores_gemma":[0.92650217,0.017282924,0.0074193003,0.037734162,0.009475775,0.001585712],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.016292991,0.0009684944,0.0012335964,0.0012691752,0.0012392735,0.0035060463,0.003166262,0.0014333902,0.0038905314],"category_scores_gemma":[0.06504876,0.0009970684,0.0006590503,0.0007229028,0.0025818164,0.0037667057,0.005012246,0.00401338,0.0015725916],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001729551,0.00072687486,0.02630228,0.0009896642,0.00017575348,0.0013966255,0.0041649835,0.029694676,0.3940074,0.25874138,0.008816888,0.2732538],"study_design_scores_gemma":[0.00030824562,0.0008598146,0.0075726816,0.00037990723,0.00015655176,0.0014651886,0.0005481401,0.16423808,0.70568603,0.043305974,0.07532215,0.00015713586],"about_ca_topic_score_codex":0.00047091144,"about_ca_topic_score_gemma":0.00042201902,"teacher_disagreement_score":0.983707,"about_ca_system_score_codex":0.0019701219,"about_ca_system_score_gemma":0.0038641123,"threshold_uncertainty_score":0.08616662},"labels":[],"label_agreement":null},{"id":"W2155388066","doi":"10.1002/stvr.210","title":"A rigorous method for test templates generation from object‐oriented specifications","year":2001,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Template; Computer science; Formal specification; Programming language; Focus (optics); Test case; White-box testing; Object-oriented programming; Formal methods; Extension (predicate logic); Specification language; Test (biology); Software engineering; Software; Software development; Software construction","score_opus":0.07291649913695267,"score_gpt":0.30554752011573033,"score_spread":0.23263102097877766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155388066","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017633445,0.000015252163,0.99673504,0.000035092544,0.0000102821905,0.00015424647,0.00002140758,0.0008371166,0.00042828353],"genre_scores_gemma":[0.05179903,0.00005852192,0.94561505,0.00007339754,0.00001716187,0.0005720021,0.00030080136,0.0005379662,0.001026079],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9828483,0.006471434,0.0011336352,0.0011561452,0.007883531,0.0005069152],"domain_scores_gemma":[0.96016157,0.025139192,0.001987503,0.007271876,0.0049925568,0.00044731543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010735871,0.0011868536,0.001089416,0.0025812823,0.00074160344,0.002095854,0.0025260383,0.0014191268,0.004325801],"category_scores_gemma":[0.046297334,0.0012014214,0.0019062845,0.0010249731,0.003292832,0.0019435106,0.0032068472,0.002244923,0.0016061929],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036051712,0.00047583948,0.002353314,0.0008101836,0.0002248219,0.0010773506,0.0017699623,0.09462225,0.06378303,0.36461055,0.0058782967,0.4640339],"study_design_scores_gemma":[0.0005520757,0.0005353374,0.00068121427,0.0003185518,0.00013362349,0.0009978855,0.00022683475,0.6496534,0.10864464,0.2006394,0.03745323,0.00016387136],"about_ca_topic_score_codex":0.0008536291,"about_ca_topic_score_gemma":0.0006829304,"teacher_disagreement_score":0.010735871,"about_ca_system_score_codex":0.00093601743,"about_ca_system_score_gemma":0.0027257216,"threshold_uncertainty_score":0.056777358},"labels":[],"label_agreement":null},{"id":"W2155775985","doi":"10.1109/ecbs.2008.49","title":"Scenario-Based Program Slicing","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Debugging; Program slicing; Slicing; Software engineering; Programming language; Microsoft Visual Studio; Agile software development; XML; Set (abstract data type); Software; Operating system; World Wide Web","score_opus":0.03759731514265637,"score_gpt":0.2735128582192871,"score_spread":0.23591554307663073,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155775985","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01013581,0.00012438964,0.98054653,0.000091175585,0.000022725653,0.0002580256,0.00038105794,0.0049376916,0.003502575],"genre_scores_gemma":[0.14684111,0.00035299873,0.84832,0.00007253532,0.000016138702,0.0004591798,0.0014401797,0.00090063777,0.0015972578],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99812454,0.00067499164,0.0002181462,0.00034218328,0.00048462927,0.0001555182],"domain_scores_gemma":[0.9960419,0.0018526975,0.0003448101,0.00096414593,0.0006356428,0.00016089277],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002495264,0.0010475847,0.0005364615,0.0012156756,0.0005787155,0.001410685,0.001162745,0.0006649284,0.006632078],"category_scores_gemma":[0.0056354785,0.0006578022,0.0013139384,0.00077789766,0.0013749859,0.0024016064,0.0021686482,0.0010355025,0.0007349604],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004943459,0.00018362021,0.0057959436,0.0011534345,0.00019641306,0.001782029,0.0036688056,0.28650445,0.053653106,0.27812314,0.0114239855,0.35702074],"study_design_scores_gemma":[0.00011254098,0.00019516941,0.0013797052,0.00033853378,0.000110742425,0.000869304,0.00046992,0.7244861,0.058908243,0.12726481,0.08573609,0.00012879001],"about_ca_topic_score_codex":0.0027484389,"about_ca_topic_score_gemma":0.0036084687,"teacher_disagreement_score":0.006632078,"about_ca_system_score_codex":0.0008667756,"about_ca_system_score_gemma":0.0015690117,"threshold_uncertainty_score":0.022186577},"labels":[],"label_agreement":null},{"id":"W2155780123","doi":"10.1145/2371401.2371407","title":"Synthesizing iterators from abstraction functions","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Programming language; Transitive closure; Tuple; Tree traversal; Reachability; Theoretical computer science; Mathematics","score_opus":0.02281374206936714,"score_gpt":0.24553140126461206,"score_spread":0.22271765919524492,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155780123","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011675695,0.000041380194,0.9820916,0.000026963377,0.000023412564,0.00006203337,0.00008926911,0.0045375796,0.0014520794],"genre_scores_gemma":[0.23833086,0.00020450748,0.7524603,0.00009143289,0.000018598868,0.0002936742,0.0008196356,0.0032219342,0.0045590643],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9982285,0.00028365047,0.00020546053,0.00028805961,0.00075279624,0.00024154605],"domain_scores_gemma":[0.99731195,0.001264989,0.00020331386,0.00074600836,0.00040564802,0.00006815834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019774418,0.0010137303,0.00088325585,0.0009235475,0.00061956816,0.002094236,0.001511494,0.0009276815,0.003371217],"category_scores_gemma":[0.004273939,0.0008508669,0.0017100412,0.0007603952,0.0012933144,0.0029937327,0.002110521,0.0019025266,0.0012250055],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007029829,0.00021719279,0.0035214773,0.0009089469,0.00019672004,0.0013998776,0.0017386123,0.12808622,0.19549818,0.40518746,0.0061944295,0.25634792],"study_design_scores_gemma":[0.00011732656,0.00027109965,0.000529683,0.00009693451,0.00019690125,0.0006493659,0.00022984619,0.41284797,0.45220962,0.08469526,0.048014857,0.00014106765],"about_ca_topic_score_codex":0.0016185534,"about_ca_topic_score_gemma":0.0017171182,"teacher_disagreement_score":0.003371217,"about_ca_system_score_codex":0.00077002164,"about_ca_system_score_gemma":0.0013439922,"threshold_uncertainty_score":0.011277795},"labels":[],"label_agreement":null},{"id":"W2155803905","doi":"10.5555/2663608.2663627","title":"BlackHorse: creating smart test cases from brittle recorded tests","year":2012,"lang":"en","type":"article","venue":"Automation of Software Test","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Blackberry (Canada); Western University","funders":"","keywords":"Computer science; Keyword-driven testing; Java; Test case; Test Management Approach; Graphical user interface testing; Test (biology); Code coverage; Manual testing; Software engineering; Test harness; Programming language; Reliability engineering; Software; Software development; User interface; Engineering; Software construction; Machine learning","score_opus":0.02530140677214751,"score_gpt":0.27447818259570933,"score_spread":0.24917677582356182,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155803905","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025766984,0.00013262352,0.92302275,0.0001907889,0.000076033146,0.0005865486,0.00050796074,0.046796456,0.0029199556],"genre_scores_gemma":[0.24504372,0.00022473939,0.7371125,0.00026915397,0.00004010126,0.0007310818,0.002462655,0.009218436,0.004897537],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99354047,0.0022279497,0.00050132355,0.0008453803,0.0024907133,0.00039424517],"domain_scores_gemma":[0.96109253,0.022258673,0.0027992786,0.00969157,0.003441748,0.00071620086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051119053,0.0015448056,0.0007074265,0.0030244945,0.00057062646,0.0027162677,0.003488221,0.0015112924,0.007121533],"category_scores_gemma":[0.041796792,0.001196742,0.0010177328,0.0010794033,0.001755952,0.003698844,0.0025485682,0.0021548818,0.0018321648],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00089886534,0.00089550193,0.015185007,0.0011335373,0.0002523043,0.0035729702,0.004249083,0.0440286,0.0778772,0.027242182,0.02639132,0.7982734],"study_design_scores_gemma":[0.00052403595,0.0012714561,0.01161928,0.0006538949,0.00016218187,0.004763575,0.0009358537,0.52614653,0.3172794,0.043202948,0.09299826,0.00044259918],"about_ca_topic_score_codex":0.0015799693,"about_ca_topic_score_gemma":0.0026749822,"teacher_disagreement_score":0.007121533,"about_ca_system_score_codex":0.0005798513,"about_ca_system_score_gemma":0.00093896285,"threshold_uncertainty_score":0.02703464},"labels":[],"label_agreement":null},{"id":"W2156213336","doi":"10.1145/2245276.2231981","title":"Symbolic execution of UML-RT State Machines","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Programming language; Unified Modeling Language; Symbolic execution; Applications of UML; Modular design; UML tool; Reachability; Finite-state machine; Code generation; Theoretical computer science; Key (lock); Software; Operating system","score_opus":0.01688462853798117,"score_gpt":0.26418489929466593,"score_spread":0.24730027075668476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2156213336","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012407004,0.00007296157,0.9796402,0.00006510716,0.00002808906,0.00007882054,0.00014674575,0.0041582277,0.003402743],"genre_scores_gemma":[0.38191646,0.0002473355,0.6111962,0.00007611175,0.000027078038,0.00033919816,0.0007508193,0.0008084047,0.0046383305],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971505,0.0011429186,0.00019543298,0.00034816345,0.0009728468,0.00019016713],"domain_scores_gemma":[0.9955959,0.0029265422,0.00038118666,0.00056025526,0.00047475658,0.00006138275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015327046,0.0008794507,0.00050302525,0.00078856904,0.00041200357,0.0011163816,0.00089989265,0.00065877876,0.0046084328],"category_scores_gemma":[0.007688743,0.00033177802,0.0009657669,0.0005185155,0.0013103505,0.0011227091,0.0011125469,0.0009530331,0.0010005159],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005468084,0.00017635275,0.0024566532,0.0007486512,0.00010203531,0.0012585131,0.001526374,0.371627,0.075566255,0.34309658,0.0036879156,0.19920686],"study_design_scores_gemma":[0.000086941036,0.00013838209,0.0004296584,0.00013322476,0.00006281496,0.00026854099,0.0000725164,0.837761,0.07389831,0.06565655,0.021445164,0.000046984398],"about_ca_topic_score_codex":0.002011296,"about_ca_topic_score_gemma":0.00174091,"teacher_disagreement_score":0.0046084328,"about_ca_system_score_codex":0.0008369456,"about_ca_system_score_gemma":0.001164439,"threshold_uncertainty_score":0.015416682},"labels":[],"label_agreement":null},{"id":"W2156296121","doi":"10.1109/issre.2007.31","title":"Using Machine Learning to Support Debugging with Tarantula","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":93,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Debugging; Computer science; Statement (logic); Ranking (information retrieval); Test (biology); Software bug; Decision tree; Fault (geology); Fault tree analysis; Machine learning; Algorithmic program debugging; Test case; Tree (set theory); Artificial intelligence; Programming language; Reliability engineering; Software; Engineering","score_opus":0.04028565877204957,"score_gpt":0.3031118466118101,"score_spread":0.26282618783976053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2156296121","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021187365,0.0000857083,0.9669476,0.00019625657,0.00002017898,0.00007932742,0.000097450895,0.010615775,0.00077031105],"genre_scores_gemma":[0.20035759,0.00008406168,0.7982815,0.0001051557,0.000019688761,0.000080458834,0.00028511783,0.00023765641,0.00054884434],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99816245,0.0007310463,0.00015538711,0.0003589726,0.00047716757,0.00011499563],"domain_scores_gemma":[0.9871875,0.008828173,0.0012245313,0.0015087496,0.0010878849,0.0001632478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029239587,0.0012890281,0.00076737977,0.0025192543,0.0006002262,0.0012327237,0.001895138,0.0009719076,0.0020115268],"category_scores_gemma":[0.018647937,0.00057348376,0.0009057835,0.0011595672,0.00060541293,0.0025326612,0.00093996135,0.0017107218,0.0009168853],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004968736,0.0005237059,0.010595613,0.00038099158,0.00018211262,0.000607656,0.00052063784,0.33046702,0.015553545,0.018767921,0.0045646364,0.61733925],"study_design_scores_gemma":[0.000021871965,0.00007980824,0.0004993915,0.000030159114,0.0000226171,0.00011899414,0.000017943135,0.9801364,0.008709285,0.008789463,0.0015503231,0.00002384158],"about_ca_topic_score_codex":0.0033919073,"about_ca_topic_score_gemma":0.005073758,"teacher_disagreement_score":0.0033919073,"about_ca_system_score_codex":0.00064333173,"about_ca_system_score_gemma":0.0012800029,"threshold_uncertainty_score":0.015463591},"labels":[],"label_agreement":null},{"id":"W2157056800","doi":"10.1145/2535838.2535857","title":"Symbolic optimization with SMT solvers","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":122,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Satisfiability modulo theories; Programming language; Symbolic execution; Satisfiability; Relation (database); Theoretical computer science; Software; Subtyping; Concolic testing; Data mining","score_opus":0.0058507943220102755,"score_gpt":0.19254675865584875,"score_spread":0.18669596433383848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157056800","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004680654,0.00055570726,0.9678385,0.0005093544,0.00011142757,0.0000815178,0.00023468923,0.0022820975,0.023705982],"genre_scores_gemma":[0.20552793,0.0013415322,0.7801227,0.00035744504,0.00020219795,0.0005148092,0.000827449,0.0010327639,0.010073119],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99835664,0.0006216952,0.00009919986,0.00016426963,0.0006202771,0.00013778002],"domain_scores_gemma":[0.9975721,0.0017597724,0.00016776979,0.0002702679,0.00018469186,0.000045354795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012702099,0.001278333,0.0010755821,0.0011701589,0.000500219,0.0015416052,0.0013706557,0.001174284,0.010235242],"category_scores_gemma":[0.006300593,0.0006491294,0.0015475228,0.0015756149,0.0016007036,0.001380855,0.0019597637,0.002155975,0.0024784151],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012767426,0.00007733975,0.0005501937,0.0006334167,0.0001064415,0.00018573322,0.00011714105,0.6318913,0.0031952602,0.25479427,0.0062181083,0.102103084],"study_design_scores_gemma":[0.00005249917,0.000024386409,0.00005589962,0.000060912298,0.000025958554,0.00004868919,0.000026355072,0.8480274,0.0022969597,0.13789687,0.011472583,0.000011470476],"about_ca_topic_score_codex":0.0023541185,"about_ca_topic_score_gemma":0.0040537124,"teacher_disagreement_score":0.010235242,"about_ca_system_score_codex":0.0011250768,"about_ca_system_score_gemma":0.0016306846,"threshold_uncertainty_score":0.034240305},"labels":[],"label_agreement":null},{"id":"W2157325856","doi":"10.1109/tase.2009.20","title":"On Testing 1-Safe Petri Nets","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Petri net; Conformance testing; Model-based testing; Finite-state machine; Programming language; Formal specification; Test case; Context (archaeology); Formal methods; Code coverage; Workflow; Formal verification; Software engineering; Software; Machine learning; Database; Operating system","score_opus":0.033475396034445624,"score_gpt":0.2690563045610275,"score_spread":0.2355809085265819,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157325856","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03796751,0.0007258578,0.955116,0.0005276499,0.0000503016,0.0001098137,0.00011225234,0.0006780353,0.0047125746],"genre_scores_gemma":[0.5320869,0.0025368603,0.45699832,0.0007036932,0.00033422728,0.0003786822,0.0007635196,0.0005987707,0.005599029],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9921069,0.0030024599,0.00047280686,0.0012156594,0.0024873733,0.00071477],"domain_scores_gemma":[0.96658593,0.028904494,0.0012732623,0.0018241614,0.0010818575,0.0003302449],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004581195,0.0021487356,0.0013128695,0.0025892179,0.0009664877,0.002084576,0.0020972607,0.0020185884,0.0033255257],"category_scores_gemma":[0.028555444,0.0010157276,0.00229452,0.00252631,0.005971903,0.0064466363,0.003695754,0.003036328,0.00063799095],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041196827,0.0001905048,0.0032186434,0.0006246698,0.00007324595,0.0007972652,0.00084060675,0.41049626,0.009202822,0.45060253,0.0011811944,0.12236031],"study_design_scores_gemma":[0.000063727195,0.00014390249,0.00031065455,0.00014558212,0.000036588503,0.00025742195,0.000079437625,0.38713777,0.00587034,0.6032069,0.0027053002,0.000042432803],"about_ca_topic_score_codex":0.0064065554,"about_ca_topic_score_gemma":0.0032155227,"teacher_disagreement_score":0.0064065554,"about_ca_system_score_codex":0.0026533282,"about_ca_system_score_gemma":0.0018999169,"threshold_uncertainty_score":0.024227977},"labels":[],"label_agreement":null},{"id":"W2157789678","doi":"10.1109/issre.2005.24","title":"Improving Statechart Testing Criteria Using Data Flow Information","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Dataflow; Unified Modeling Language; Tree (set theory); Data flow diagram; Control flow; Data-flow analysis; Test case; Data mining; Path (computing); Database; Machine learning; Programming language","score_opus":0.07316659845003702,"score_gpt":0.29757673137721696,"score_spread":0.22441013292717993,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157789678","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19550219,0.00034684504,0.7957939,0.00040653694,0.000022657723,0.0004761578,0.00027063323,0.0029442068,0.004236869],"genre_scores_gemma":[0.70131475,0.0001220886,0.2970784,0.00006970578,0.000014081351,0.0002408632,0.0004934753,0.00022229653,0.00044437352],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98870987,0.0039872634,0.000833743,0.00073734066,0.005277545,0.00045414115],"domain_scores_gemma":[0.9383524,0.044856913,0.004766091,0.0024364702,0.009033319,0.0005547761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00681961,0.0016267954,0.0011031474,0.007980781,0.0005924781,0.0023679514,0.0015462913,0.0011092653,0.0016659732],"category_scores_gemma":[0.057846162,0.00049217837,0.0009771009,0.0023157925,0.00092966965,0.0036311168,0.0013111576,0.00069111754,0.00027484223],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006199691,0.00072361826,0.040638354,0.00085640134,0.00015419348,0.00079661806,0.0012093135,0.2526793,0.05011711,0.031673342,0.0017908956,0.6187409],"study_design_scores_gemma":[0.00009432337,0.0006408718,0.008828444,0.00023216306,0.00012806854,0.00049053674,0.0003184914,0.9126708,0.056170315,0.016449487,0.003871758,0.00010479679],"about_ca_topic_score_codex":0.0038580454,"about_ca_topic_score_gemma":0.0052073374,"teacher_disagreement_score":0.007980781,"about_ca_system_score_codex":0.0016669442,"about_ca_system_score_gemma":0.0024038088,"threshold_uncertainty_score":0.036065936},"labels":[],"label_agreement":null},{"id":"W2157827846","doi":"10.1145/1967677.1967693","title":"Software debugging and testing using the abstract diagnosis theory","year":2011,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Observability; Debugging; Dependency (UML); Controllability; Program slicing; Variable (mathematics); Context (archaeology); Set (abstract data type); Algorithm; Theoretical computer science; Programming language; Mathematics","score_opus":0.12479484147938143,"score_gpt":0.27631308476431243,"score_spread":0.15151824328493102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157827846","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032180117,0.00032038076,0.99427265,0.00031834806,0.00003237777,0.000054739223,0.000047581216,0.00022018795,0.0015158359],"genre_scores_gemma":[0.22798774,0.0015238263,0.7665526,0.00042592484,0.00030171472,0.00037642152,0.0004350843,0.000117897565,0.0022787508],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9952833,0.0015658046,0.0003394174,0.0008685291,0.0016292036,0.0003137039],"domain_scores_gemma":[0.9912322,0.0067778802,0.0005679194,0.0007765099,0.00049659394,0.00014883504],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027099808,0.0017384181,0.0015086232,0.0029496846,0.00089147285,0.0026994017,0.001789856,0.001760645,0.0033796658],"category_scores_gemma":[0.010917738,0.0007918869,0.0027772645,0.0018122697,0.004341347,0.005986995,0.0028380598,0.003802406,0.00056694215],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015555181,0.00012159946,0.0010447638,0.00057459,0.000108314336,0.00033935727,0.0003023182,0.25892746,0.0048778895,0.60330135,0.0020527388,0.12819415],"study_design_scores_gemma":[0.00005711504,0.000085015825,0.00019976862,0.00008624649,0.00004323296,0.00013368785,0.000050725295,0.4927818,0.0020542834,0.500015,0.0044668945,0.000026236725],"about_ca_topic_score_codex":0.0031459793,"about_ca_topic_score_gemma":0.0018393575,"teacher_disagreement_score":0.0033796658,"about_ca_system_score_codex":0.002216937,"about_ca_system_score_gemma":0.0022741498,"threshold_uncertainty_score":0.016085088},"labels":[],"label_agreement":null},{"id":"W2158798798","doi":"10.1145/1368088.1368118","title":"Calysto","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":146,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Scalability; Static analysis; Source lines of code; Automation; Static program analysis; Context (archaeology); Software; Software bug; Model checking; Code (set theory); Computer engineering; Programming language; Software development; Database; Set (abstract data type)","score_opus":0.03481987681072068,"score_gpt":0.23643006246479897,"score_spread":0.2016101856540783,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2158798798","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026042784,0.0029894656,0.28490058,0.002108302,0.0013806805,0.00084726885,0.01994323,0.51401055,0.14777721],"genre_scores_gemma":[0.27495834,0.0022875816,0.41318518,0.0036466918,0.00043814024,0.0012894678,0.08133406,0.08095688,0.14190371],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99827766,0.00019464335,0.00008530222,0.00039215566,0.0008404047,0.00020984036],"domain_scores_gemma":[0.9970528,0.000957906,0.0002754534,0.0007402353,0.00080538716,0.00016816352],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0011781451,0.0012641593,0.0008444654,0.0015053414,0.0008443953,0.0024445073,0.0032019995,0.0013749478,0.04899959],"category_scores_gemma":[0.006284682,0.00084720284,0.0009143074,0.0010387048,0.0009883365,0.003783002,0.0022911439,0.0023427333,0.017951744],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020752437,0.00022159978,0.004153631,0.0014800585,0.00013019082,0.0005784253,0.0003975205,0.007527835,0.028978549,0.051526405,0.57225794,0.3306726],"study_design_scores_gemma":[0.00067786063,0.00037091857,0.0034077512,0.00030104132,0.000118029246,0.0010173976,0.00012185885,0.0872624,0.047238566,0.03809023,0.8212126,0.0001813753],"about_ca_topic_score_codex":0.004978954,"about_ca_topic_score_gemma":0.007093893,"teacher_disagreement_score":0.9510004,"about_ca_system_score_codex":0.0012489038,"about_ca_system_score_gemma":0.00208229,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2160197556","doi":"10.1109/tc.2006.80","title":"Optimizing the length of checking sequences","year":2006,"lang":"en","type":"article","venue":"IEEE Transactions on Computers","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":93,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Engineering and Physical Sciences Research Council; Natural Sciences and Engineering Research Council of Canada; Leverhulme Trust","keywords":"Sequence (biology); State (computer science); Set (abstract data type); Algorithm; Finite-state machine; Mathematics; Computer science; Combinatorics","score_opus":0.021365074338475912,"score_gpt":0.24324589287024515,"score_spread":0.22188081853176922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2160197556","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28782773,0.001356049,0.6954608,0.00055577856,0.00013565947,0.00052717194,0.0005282341,0.0037151594,0.009893345],"genre_scores_gemma":[0.6097521,0.00040354277,0.38420922,0.00015029933,0.000041666703,0.00030187247,0.0008898994,0.00077924493,0.0034721394],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99746895,0.0006963806,0.00026979318,0.00062080333,0.0006153698,0.00032870306],"domain_scores_gemma":[0.9915406,0.004781183,0.0012070613,0.00075175747,0.0013312354,0.0003882332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018161122,0.0012442678,0.00097400666,0.001257092,0.000626492,0.0012318082,0.0016268782,0.00093704986,0.0033062086],"category_scores_gemma":[0.011243792,0.0007815522,0.00068664114,0.001228095,0.00073960604,0.0020924222,0.0009561805,0.0011486149,0.0009924689],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001706514,0.0005095878,0.008256962,0.00063322205,0.00014010587,0.0003555253,0.00034781426,0.6055328,0.07574732,0.010246743,0.0030395584,0.2934838],"study_design_scores_gemma":[0.00014466647,0.0009740488,0.002371745,0.00008764779,0.00015110885,0.0003784322,0.00024144087,0.9167105,0.060696386,0.012903042,0.005295115,0.00004583855],"about_ca_topic_score_codex":0.0016875523,"about_ca_topic_score_gemma":0.0030612957,"teacher_disagreement_score":0.0033062086,"about_ca_system_score_codex":0.0013627373,"about_ca_system_score_gemma":0.0025018717,"threshold_uncertainty_score":0.011060357},"labels":[],"label_agreement":null},{"id":"W2160450788","doi":"10.1145/337180.337825","title":"Workshop on standard exchange format (WoSEF) (workshop session)","year":2000,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of Toronto","funders":"","keywords":"Session (web analytics); Computer science; World Wide Web","score_opus":0.031036539898250406,"score_gpt":0.2906063153049976,"score_spread":0.2595697754067472,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2160450788","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018913884,0.007574819,0.81415987,0.013995994,0.0127279945,0.0014669369,0.0046455325,0.012274865,0.114240125],"genre_scores_gemma":[0.09632116,0.013534352,0.49384573,0.0037269357,0.004948588,0.001867671,0.03343685,0.008011577,0.3443071],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9956737,0.0019105185,0.00031901014,0.0004515694,0.0011733731,0.00047175834],"domain_scores_gemma":[0.9879673,0.003702844,0.0003613889,0.0033885187,0.0035372349,0.0010427502],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014559663,0.0013418231,0.0010715261,0.0025309417,0.0011312013,0.0050709066,0.0024340693,0.0028386037,0.10923353],"category_scores_gemma":[0.021020021,0.0007203288,0.0011001219,0.0028705047,0.0010041805,0.010917018,0.003258512,0.0026413929,0.044016644],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007404696,0.00039959818,0.0011913569,0.0004832702,0.000056042794,0.0003278379,0.00046263714,0.0015852978,0.008032489,0.080720015,0.308167,0.59783405],"study_design_scores_gemma":[0.00015004052,0.0004628609,0.0015498332,0.00070705794,0.000054411656,0.00080865144,0.00044366595,0.008639797,0.012545243,0.048329838,0.9262244,0.0000842659],"about_ca_topic_score_codex":0.0021255042,"about_ca_topic_score_gemma":0.0016577148,"teacher_disagreement_score":0.10923353,"about_ca_system_score_codex":0.0007680126,"about_ca_system_score_gemma":0.0021075536,"threshold_uncertainty_score":0.36542255},"labels":[],"label_agreement":null},{"id":"W2160917074","doi":"10.1142/s0218194013500113","title":"USING TEST ORACLES AND FORMAL SPECIFICATIONS WITH TEST-DRIVEN DEVELOPMENT","year":2013,"lang":"en","type":"article","venue":"International Journal of Software Engineering and Knowledge Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Computer science; Documentation; Software engineering; Formal specification; Programming language; Test (biology); Formal methods; Test case; Notation; Test Management Approach; Test-driven development; Test suite; Process (computing); Software development; Software; Software construction","score_opus":0.026056357144596323,"score_gpt":0.2394550303039406,"score_spread":0.2133986731593443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2160917074","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023670082,0.00018168674,0.9938937,0.00042632854,0.000034620305,0.000083289415,0.000025825688,0.0010149805,0.0019725822],"genre_scores_gemma":[0.07969549,0.00044111902,0.9179252,0.00025443142,0.00004177919,0.00024255636,0.00015132234,0.00036470895,0.00088338135],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.97040135,0.016424764,0.0027685103,0.0015669639,0.008165366,0.00067315326],"domain_scores_gemma":[0.9138933,0.06509438,0.0038385645,0.012455282,0.003985116,0.00073346694],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025089879,0.0017807676,0.0009999031,0.0031759585,0.0007550454,0.005668795,0.0028345427,0.0025212984,0.0027065894],"category_scores_gemma":[0.09047998,0.0014372913,0.001837975,0.0017056341,0.0072607426,0.009460353,0.0060215862,0.00408357,0.0007780004],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016136118,0.000247668,0.0027340525,0.00080271834,0.00009984547,0.0011563835,0.002239316,0.06677162,0.0047505703,0.7141232,0.001853768,0.20505953],"study_design_scores_gemma":[0.0002746809,0.00037068807,0.0005994755,0.0011595918,0.0001509932,0.0019409668,0.00044304944,0.26749444,0.020844769,0.6433297,0.063124485,0.0002670949],"about_ca_topic_score_codex":0.0022173233,"about_ca_topic_score_gemma":0.0016969766,"teacher_disagreement_score":0.025089879,"about_ca_system_score_codex":0.0016617277,"about_ca_system_score_gemma":0.002621966,"threshold_uncertainty_score":0.13268954},"labels":[],"label_agreement":null},{"id":"W2162032875","doi":"10.1109/real.1998.739748","title":"Timed test cases generation based on state characterization technique","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":112,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Université de Montréal","funders":"","keywords":"Computer science; Nondeterministic algorithm; Correctness; Test suite; Automaton; Timed automaton; Finite-state machine; Set (abstract data type); System under test; State (computer science); Granularity; Algorithm; Test case; Programming language; Theoretical computer science","score_opus":0.0420348950945121,"score_gpt":0.24113334313612084,"score_spread":0.19909844804160873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2162032875","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02302865,0.000025389534,0.9731506,0.000049641603,0.000016482916,0.00019293331,0.00012598776,0.0023789573,0.0010312799],"genre_scores_gemma":[0.39174715,0.00007184843,0.60491836,0.0000717466,0.000016672084,0.0006129507,0.0008972094,0.0004378239,0.0012261886],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978988,0.00062841346,0.0002273236,0.00035897552,0.00073703437,0.00014942279],"domain_scores_gemma":[0.9946243,0.0031914597,0.00051592104,0.00079306174,0.0007826265,0.000092524206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00097360637,0.0007296949,0.00048085817,0.0013349055,0.00026463266,0.00067884265,0.0009432107,0.00067695195,0.0024467935],"category_scores_gemma":[0.0075672506,0.00035156545,0.00077674614,0.000679184,0.0006877309,0.00089505344,0.00069648615,0.0007471022,0.0004667101],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006571556,0.00048454572,0.007110064,0.0004606148,0.00011602217,0.0019210441,0.00084022543,0.26278806,0.1468424,0.08649131,0.0050226133,0.48726594],"study_design_scores_gemma":[0.00012815745,0.00022538414,0.00080207334,0.000047702146,0.000066154906,0.0006566898,0.000056180656,0.8705558,0.100202784,0.019938534,0.007271823,0.00004869655],"about_ca_topic_score_codex":0.00125826,"about_ca_topic_score_gemma":0.00086739525,"teacher_disagreement_score":0.0024467935,"about_ca_system_score_codex":0.00054666767,"about_ca_system_score_gemma":0.0007402841,"threshold_uncertainty_score":0.008185387},"labels":[],"label_agreement":null},{"id":"W2163289685","doi":"10.1109/icsm.2002.1167775","title":"Automating impact analysis and regression test selection based on UML designs","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":131,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Unified Modeling Language; Regression testing; Selection (genetic algorithm); Test (biology); Regression analysis; Data mining; Programming language; Software engineering; Artificial intelligence; Machine learning; Software; Software development","score_opus":0.02588388994970477,"score_gpt":0.3075896533459878,"score_spread":0.28170576339628306,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2163289685","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09070243,0.0002322759,0.8994449,0.0001667967,0.00002516777,0.00034947402,0.00009385286,0.007187383,0.0017977026],"genre_scores_gemma":[0.3857545,0.00012025069,0.6123359,0.00008640319,0.000029056353,0.00037314522,0.0002954872,0.00032162637,0.00068367535],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9900649,0.0050327755,0.0005483393,0.0006850033,0.0032513149,0.00041766127],"domain_scores_gemma":[0.9587304,0.027499162,0.0048141,0.004260981,0.0042671454,0.00042827844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064208615,0.0012770863,0.0009109926,0.004381758,0.00039466628,0.0011996329,0.0012290347,0.00094513403,0.0013878781],"category_scores_gemma":[0.03311447,0.00043713453,0.0007739286,0.0014072238,0.0005836777,0.0011050712,0.0008665691,0.0008373185,0.00049345184],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031555112,0.00051265716,0.025670148,0.00036990517,0.00016467061,0.00061483524,0.0006499781,0.055965636,0.054393332,0.0066229333,0.0016115167,0.85310894],"study_design_scores_gemma":[0.00022071852,0.0008648771,0.016519794,0.00016492362,0.00025407574,0.0010120887,0.00022349026,0.8691687,0.08576689,0.0179903,0.007713232,0.00010098698],"about_ca_topic_score_codex":0.0016361652,"about_ca_topic_score_gemma":0.0018792168,"teacher_disagreement_score":0.0064208615,"about_ca_system_score_codex":0.0006673203,"about_ca_system_score_gemma":0.0013700476,"threshold_uncertainty_score":0.033957183},"labels":[],"label_agreement":null},{"id":"W2163844940","doi":"10.1109/ccece.2011.6030533","title":"Integration testing object-oriented software systems: An experiment-driven research approach","year":2011,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Integration testing; Unit testing; Computer science; Class (philosophy); Process (computing); White-box testing; System integration; Software engineering; System integration testing; Test strategy; Software; Systems engineering; Programming language; Artificial intelligence; Software construction; Software development; Engineering; Database","score_opus":0.24943487068187783,"score_gpt":0.35682480079505396,"score_spread":0.10738993011317613,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2163844940","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21093819,0.0010439092,0.7688197,0.001816614,0.00024484654,0.0084711015,0.00034306553,0.0004659888,0.0078566205],"genre_scores_gemma":[0.43501955,0.0013783005,0.5470854,0.0010106731,0.00018537634,0.013533679,0.00028523745,0.00011116657,0.0013905097],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9406347,0.050492432,0.002031543,0.0022224109,0.0041145356,0.0005043151],"domain_scores_gemma":[0.7597368,0.21776316,0.0047070705,0.0100893285,0.0065721706,0.0011314533],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.054169383,0.001049863,0.0011875813,0.0014273116,0.0008852375,0.002411615,0.0029908107,0.0015579929,0.0020086188],"category_scores_gemma":[0.08130907,0.00074012356,0.0010189619,0.00143864,0.003797624,0.0035717767,0.001529393,0.0017860245,0.00036097958],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.011126862,0.03233083,0.044388916,0.008509294,0.0014025526,0.000499071,0.014942326,0.053514082,0.090699054,0.26134124,0.004058953,0.47718686],"study_design_scores_gemma":[0.010688708,0.075801626,0.026887912,0.002299427,0.0023104872,0.00091143884,0.011353227,0.24245611,0.13719352,0.43892688,0.050275322,0.0008953304],"about_ca_topic_score_codex":0.000919297,"about_ca_topic_score_gemma":0.0008238428,"teacher_disagreement_score":0.054169383,"about_ca_system_score_codex":0.002631202,"about_ca_system_score_gemma":0.002201466,"threshold_uncertainty_score":0.28647852},"labels":[],"label_agreement":null},{"id":"W2164188174","doi":"10.1109/compsac.2005.151","title":"Testing the Semantics of W3C XML Schema","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; XML Schema Editor; XML Schema (W3C); RELAX NG; Document Structure Description; XML validation; Streaming XML; Efficient XML Interchange; XML; Programming language; Information retrieval; World Wide Web; XML Signature","score_opus":0.031104465118838128,"score_gpt":0.2450003664756878,"score_spread":0.21389590135684966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164188174","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41362756,0.00018428736,0.5678853,0.0008594529,0.00012687586,0.00065574876,0.0010150088,0.010093123,0.0055527794],"genre_scores_gemma":[0.70126164,0.00019809182,0.29330766,0.00040968892,0.000026448615,0.0004531791,0.0019140093,0.0012335577,0.0011957362],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9799635,0.007043638,0.0017917451,0.0016648385,0.008911621,0.00062455976],"domain_scores_gemma":[0.961508,0.019806989,0.0029991951,0.008218038,0.007091567,0.00037625997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01160214,0.0008074209,0.00047854485,0.0018583429,0.0008287497,0.0023247965,0.0020581882,0.0020573167,0.0010227837],"category_scores_gemma":[0.05505659,0.00052917824,0.0011027697,0.0011872408,0.001865005,0.0055459286,0.0015867294,0.0015189013,0.00038949467],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017821083,0.0023470283,0.054644868,0.0013028058,0.00081225386,0.002958141,0.0044627846,0.11235923,0.28686252,0.20915414,0.009621781,0.3136924],"study_design_scores_gemma":[0.0002601959,0.00084959,0.005195677,0.00019709629,0.0002081862,0.0010975687,0.00082179834,0.5521052,0.36791146,0.05528031,0.015937973,0.00013490497],"about_ca_topic_score_codex":0.004086581,"about_ca_topic_score_gemma":0.0033299162,"teacher_disagreement_score":0.01160214,"about_ca_system_score_codex":0.001379283,"about_ca_system_score_gemma":0.0029814,"threshold_uncertainty_score":0.06135869},"labels":[],"label_agreement":null},{"id":"W2164209468","doi":"10.1109/enabl.2001.953386","title":"SCENTOR: scenario-based testing of e-business applications","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Software engineering; Extreme programming; Schedule; Pace; Task (project management); Business requirements; Black-box testing; Software development; Software; Business process; Systems engineering; Software development process; Software construction; Programming language; Operating system; Engineering","score_opus":0.05492985228936814,"score_gpt":0.24851425001435307,"score_spread":0.19358439772498492,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164209468","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08955314,0.00020814379,0.8711857,0.00055672904,0.000070919275,0.0003211278,0.0005261006,0.02778369,0.009794438],"genre_scores_gemma":[0.6142267,0.00021963954,0.38012436,0.0001773512,0.000026150201,0.00025376736,0.0011700521,0.0008731906,0.002928759],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99794,0.0012723054,0.00011295357,0.00014711423,0.0004191606,0.00010859054],"domain_scores_gemma":[0.9941737,0.0041082473,0.00028097216,0.00081696676,0.00039064966,0.00022949914],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025112687,0.0007640073,0.00026846468,0.0009573988,0.00026807922,0.00094099226,0.0017528064,0.001098574,0.0050496887],"category_scores_gemma":[0.010197058,0.0003100469,0.00037107663,0.0005347734,0.00080177566,0.0019599472,0.0011475579,0.0006758715,0.000737896],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016121762,0.001036579,0.023401044,0.0009320701,0.00022480729,0.004866292,0.0021471803,0.16856472,0.06425584,0.08120276,0.024562092,0.62719446],"study_design_scores_gemma":[0.00028399122,0.0008864513,0.0062317196,0.00016516744,0.00005940331,0.0020730246,0.00043919805,0.87057066,0.05820062,0.03534405,0.025658838,0.00008676769],"about_ca_topic_score_codex":0.0010816514,"about_ca_topic_score_gemma":0.0010612372,"teacher_disagreement_score":0.0050496887,"about_ca_system_score_codex":0.0002731588,"about_ca_system_score_gemma":0.0004327387,"threshold_uncertainty_score":0.01689285},"labels":[],"label_agreement":null},{"id":"W2164551036","doi":"10.1109/sew.2003.1270746","title":"Software verification and validation within the (rational) unified process","year":2004,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Computer science; Software verification; Verification and validation; Process (computing); Software engineering; Unified Process; Functional verification; Verification; Formal verification; Software; Programming language; Unified Modeling Language; Software construction; Software development; Engineering","score_opus":0.024221883433120418,"score_gpt":0.2688568003011636,"score_spread":0.24463491686804317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164551036","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0061974637,0.0006199113,0.9851692,0.00040756425,0.000058254715,0.00015424757,0.000017341503,0.0006268936,0.0067491066],"genre_scores_gemma":[0.21697779,0.001398527,0.7775344,0.00023838383,0.00010590169,0.00049901387,0.00012169707,0.00017027085,0.0029539664],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.96347857,0.018676378,0.0021859636,0.0019403503,0.0119905,0.0017283099],"domain_scores_gemma":[0.9655718,0.016908133,0.0033820844,0.008460704,0.0052342,0.0004430598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027404636,0.0013258903,0.0013897113,0.0023245267,0.0015942581,0.0073207757,0.0030543515,0.0031269449,0.0013360706],"category_scores_gemma":[0.04158225,0.0009077689,0.0018623705,0.0022023942,0.008876551,0.0071577984,0.0029135023,0.0030299372,0.0010300968],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000098664765,0.00006025001,0.0007134334,0.00024421696,0.000051806626,0.00023045587,0.0011133797,0.028318679,0.002320518,0.9195622,0.0007209202,0.046565454],"study_design_scores_gemma":[0.000193343,0.00055817614,0.00076004927,0.00065133825,0.00024040432,0.00055399083,0.00042864916,0.2636424,0.019696804,0.6562926,0.056800067,0.00018216512],"about_ca_topic_score_codex":0.005283243,"about_ca_topic_score_gemma":0.0032405127,"teacher_disagreement_score":0.027404636,"about_ca_system_score_codex":0.002095532,"about_ca_system_score_gemma":0.006078856,"threshold_uncertainty_score":0.14493132},"labels":[],"label_agreement":null},{"id":"W2164617101","doi":"10.1145/1082983.1083247","title":"Dynamic analysis of java applications for multithreaded antipatterns","year":2005,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Java; Tracing; Programming language; TRACE (psycholinguistics); Scalability; Model checking; Formal methods; Java Modeling Language; Formal specification; Software engineering; Eclipse; Java annotation; Real time Java; Operating system","score_opus":0.018557714815639687,"score_gpt":0.2824922130252833,"score_spread":0.26393449820964365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164617101","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7361228,0.00038310522,0.2530608,0.00022628182,0.000044793524,0.00013561931,0.00016660093,0.0064630015,0.00339704],"genre_scores_gemma":[0.9497612,0.00011111024,0.048468415,0.00005655619,0.000010427538,0.000077570185,0.00020452132,0.00034737732,0.0009628141],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99872583,0.00025787743,0.00009544568,0.00020805029,0.0005380953,0.00017462843],"domain_scores_gemma":[0.9959345,0.0020430228,0.00058605353,0.00077071594,0.00056440843,0.00010129734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013217272,0.00050068746,0.0004396371,0.00108259,0.000474467,0.0010414872,0.0008412095,0.0006586893,0.0011722818],"category_scores_gemma":[0.0054041236,0.00043398776,0.00068866415,0.000674001,0.00078719546,0.0013178625,0.0008655154,0.0006048835,0.00014692248],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00091145776,0.0007046972,0.0823006,0.00076315086,0.00017497878,0.0016261358,0.0010626398,0.15404734,0.45687953,0.031770922,0.0021220937,0.26763642],"study_design_scores_gemma":[0.000092212926,0.00026601972,0.018916886,0.000054553027,0.00009498442,0.00048705886,0.00014117171,0.8435475,0.11522724,0.01709881,0.0040312787,0.000042287986],"about_ca_topic_score_codex":0.002524782,"about_ca_topic_score_gemma":0.0036301075,"teacher_disagreement_score":0.002524782,"about_ca_system_score_codex":0.0007259028,"about_ca_system_score_gemma":0.0011617114,"threshold_uncertainty_score":0.006990075},"labels":[],"label_agreement":null},{"id":"W2164968206","doi":"10.1109/icst.2012.111","title":"Generating Checking Sequences for Nondeterministic Finite State Machines","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Nondeterministic algorithm; Finite-state machine; Computer science; Sequence (biology); Model checking; Deterministic finite automaton; Theoretical computer science; Generalization; State (computer science); Automaton; Class (philosophy); Algorithm; Fault coverage; Programming language; Mathematics; Artificial intelligence","score_opus":0.048602551238794436,"score_gpt":0.3108628382386156,"score_spread":0.2622602869998212,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164968206","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04878637,0.00005543914,0.94614524,0.00009688012,0.000039035407,0.00023476168,0.00025185433,0.0031544412,0.0012359893],"genre_scores_gemma":[0.30724657,0.00009530558,0.6883019,0.00010744484,0.00003367663,0.000594659,0.0013866207,0.0006532462,0.0015805989],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99589217,0.0013598839,0.0004714602,0.0008165375,0.0011813951,0.00027862162],"domain_scores_gemma":[0.9729283,0.019237734,0.0016528589,0.0037107384,0.0021514348,0.00031886908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031577083,0.001031506,0.0006022126,0.0014434597,0.0007347808,0.0008671238,0.0013898821,0.00092302583,0.0024281484],"category_scores_gemma":[0.019886132,0.0007600344,0.0012461109,0.000689468,0.0014563543,0.001923635,0.0012662286,0.0011426694,0.0007520474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011850634,0.00053476664,0.012283756,0.0012425531,0.00016357517,0.0016859202,0.0018995613,0.27399367,0.08521384,0.22880933,0.004505045,0.38848293],"study_design_scores_gemma":[0.00014794513,0.00037132142,0.0009501772,0.00018487059,0.000098522825,0.00049721694,0.00011275356,0.6307001,0.16324918,0.1940129,0.009587604,0.00008740516],"about_ca_topic_score_codex":0.0008313717,"about_ca_topic_score_gemma":0.0011878648,"teacher_disagreement_score":0.0031577083,"about_ca_system_score_codex":0.00073821394,"about_ca_system_score_gemma":0.0019723517,"threshold_uncertainty_score":0.016699731},"labels":[],"label_agreement":null},{"id":"W2165679316","doi":"10.1109/icst.2008.18","title":"A Three-Tiered Testing Strategy for Cookies","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Scalability; Test suite; Random testing; Grammar; Suite; Reliability (semiconductor); Artificial intelligence; Test case; Machine learning; Database","score_opus":0.15301318136428213,"score_gpt":0.2990189896137123,"score_spread":0.14600580824943019,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2165679316","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017097808,0.00006623303,0.9773013,0.0003144638,0.00001932337,0.00035287952,0.000052343494,0.00163587,0.0031598494],"genre_scores_gemma":[0.20812583,0.00007909424,0.78780943,0.00033807484,0.000014300928,0.00033028665,0.00021071653,0.0004296035,0.0026627758],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9862251,0.0047827545,0.0013077587,0.0018027537,0.005242879,0.0006387481],"domain_scores_gemma":[0.97892994,0.008604442,0.001136839,0.0053118467,0.005494643,0.00052234664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008651663,0.001157816,0.00072813814,0.0026646787,0.0011253286,0.002624793,0.0033524656,0.0017707329,0.0022173],"category_scores_gemma":[0.022822024,0.0006937079,0.001142696,0.0008064428,0.004207047,0.0048935255,0.0034729245,0.0022211443,0.00075149216],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035669396,0.00059373776,0.011150467,0.0004881118,0.00013145873,0.0013190173,0.0035453753,0.048966687,0.09639368,0.40067992,0.0040868907,0.43228793],"study_design_scores_gemma":[0.000106524734,0.0010867894,0.0031033242,0.0003798664,0.0001992762,0.003506962,0.0011796128,0.49138474,0.15598977,0.29554194,0.04720766,0.00031348708],"about_ca_topic_score_codex":0.0031748551,"about_ca_topic_score_gemma":0.0045654504,"teacher_disagreement_score":0.008651663,"about_ca_system_score_codex":0.0014623335,"about_ca_system_score_gemma":0.0033671951,"threshold_uncertainty_score":0.04575491},"labels":[],"label_agreement":null},{"id":"W2166103344","doi":"10.1109/ccece.1999.807212","title":"Formal test requirements for component interactions","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Integration testing; Concurrency; Unit testing; Component (thermodynamics); Model-based testing; Formal methods; Conformance testing; Test case; Formal specification; Test (biology); Software engineering; Programming language; Reliability engineering; Software; Machine learning; Engineering; Standardization","score_opus":0.05891314489628981,"score_gpt":0.32213162065631656,"score_spread":0.26321847576002677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166103344","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016884053,0.00019497359,0.9730407,0.00095156615,0.000087023705,0.00027503158,0.00028273987,0.0006050941,0.0076788017],"genre_scores_gemma":[0.55935395,0.00080200494,0.42864424,0.0010412482,0.00038033046,0.002763443,0.0020181513,0.0005136977,0.0044830022],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9727113,0.008521236,0.0032389755,0.0014896701,0.012465049,0.0015737031],"domain_scores_gemma":[0.9262185,0.05085659,0.0046430207,0.006710486,0.0106353415,0.0009359846],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010976787,0.0017534948,0.0010958299,0.0021272323,0.0010606778,0.003281629,0.0024553044,0.004000314,0.0036580786],"category_scores_gemma":[0.062276214,0.0010509198,0.0022130166,0.0009360393,0.0041475534,0.0056443866,0.003019904,0.0033303797,0.0012874476],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017191205,0.00026195365,0.0018631847,0.0004562792,0.000069039124,0.0013074776,0.0012266291,0.086524755,0.013424374,0.8636883,0.0039024802,0.027103627],"study_design_scores_gemma":[0.00026420428,0.00027145917,0.00057780516,0.0002781692,0.00009092579,0.0012432799,0.00034147984,0.29706156,0.0152500775,0.66084665,0.023681104,0.000093301904],"about_ca_topic_score_codex":0.0018046416,"about_ca_topic_score_gemma":0.0011665963,"teacher_disagreement_score":0.010976787,"about_ca_system_score_codex":0.0015288529,"about_ca_system_score_gemma":0.002678986,"threshold_uncertainty_score":0.058051527},"labels":[],"label_agreement":null},{"id":"W2166881673","doi":"10.1109/snpd.2009.55","title":"A Tool Suite for Java Program Tracing and Feature Location","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Algoma University","funders":"Algoma University","keywords":"Suite; Java; Debugging; Computer science; Programmer; Tracing; Programming language; Feature (linguistics); TRACE (psycholinguistics); Source code; Call graph; Focus (optics); Operating system","score_opus":0.016399273823345044,"score_gpt":0.291218842187196,"score_spread":0.27481956836385096,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166881673","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031729222,0.00034619126,0.75109774,0.00012824453,0.00010104299,0.00045512992,0.0036517691,0.23835775,0.0026892081],"genre_scores_gemma":[0.042454217,0.000998308,0.8766185,0.0003787457,0.000101098296,0.002502355,0.02770945,0.041508056,0.0077292724],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99668664,0.0006460275,0.0006906417,0.00042475047,0.0013439333,0.00020807565],"domain_scores_gemma":[0.9926864,0.004160499,0.0005827993,0.0013329948,0.0009885262,0.000248868],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031990937,0.0027159553,0.001401852,0.0047986475,0.00067188015,0.0018242153,0.002992998,0.0016338488,0.009123844],"category_scores_gemma":[0.013021483,0.0021687895,0.0020681787,0.0029850262,0.0005888113,0.0031919111,0.0024713615,0.002565096,0.006838137],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058310694,0.00072673894,0.0052343123,0.0024196808,0.00043523198,0.0019113069,0.0009799782,0.018228715,0.036459345,0.017968059,0.15125282,0.7638007],"study_design_scores_gemma":[0.001188553,0.0006309363,0.009045255,0.0011161882,0.00037430058,0.007098359,0.00033121498,0.27720872,0.10479316,0.053289283,0.544307,0.0006170815],"about_ca_topic_score_codex":0.0023500447,"about_ca_topic_score_gemma":0.002300629,"teacher_disagreement_score":0.009123844,"about_ca_system_score_codex":0.0005531998,"about_ca_system_score_gemma":0.0017409264,"threshold_uncertainty_score":0.030522287},"labels":[],"label_agreement":null},{"id":"W2166906534","doi":"10.1109/infcom.1993.253243","title":"A unified approach to protocol test sequence generation","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":78,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Executable; Extended finite-state machine; Computer science; Constraint satisfaction problem; Sequence (biology); Theoretical computer science; Control flow; Finite-state machine; Control flow graph; Constraint (computer-aided design); Algorithm; Data flow diagram; Construct (python library); Graph; Data-flow analysis; Data mining; Programming language; Artificial intelligence; Mathematics; Database","score_opus":0.14543768759567544,"score_gpt":0.30963796361751794,"score_spread":0.1642002760218425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166906534","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000361794,0.000037280362,0.9977458,0.00004421178,0.000015895337,0.000099907105,0.000026529155,0.0007050686,0.0009634715],"genre_scores_gemma":[0.02264113,0.00015111975,0.9747131,0.00009902834,0.000034428525,0.0005449279,0.00026295823,0.00024126262,0.00131209],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99324495,0.0020241593,0.00058719283,0.00097501924,0.0028095073,0.0003592719],"domain_scores_gemma":[0.99463636,0.0015724147,0.00028273946,0.0017900455,0.0015822939,0.00013606739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037769058,0.0013224523,0.0013160005,0.0034082264,0.0009908467,0.0032077634,0.0031617924,0.0018279473,0.006288102],"category_scores_gemma":[0.0100380955,0.0010869203,0.002177529,0.0023808393,0.0018990807,0.003586568,0.003080431,0.0029723225,0.0021135472],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012762076,0.00022390003,0.0006326088,0.000400063,0.00012834674,0.00036337363,0.00040121615,0.080775216,0.013384338,0.41525903,0.008221441,0.4800828],"study_design_scores_gemma":[0.00010950373,0.00023441472,0.00026423694,0.00020022542,0.00012944276,0.0005880206,0.0001359189,0.7241112,0.017738588,0.20400855,0.052388854,0.00009102764],"about_ca_topic_score_codex":0.0017902751,"about_ca_topic_score_gemma":0.0020438402,"teacher_disagreement_score":0.006288102,"about_ca_system_score_codex":0.0012000307,"about_ca_system_score_gemma":0.0034578375,"threshold_uncertainty_score":0.02103585},"labels":[],"label_agreement":null},{"id":"W2167505893","doi":"10.1109/tr.2009.2034288","title":"A Novel Evolutionary Approach for Adaptive Random Testing","year":2009,"lang":"en","type":"article","venue":"IEEE Transactions on Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Random testing; Sobol sequence; Computer science; Evolutionary algorithm; Test strategy; Generality; Orthogonal array testing; Mathematical optimization; Random search; Algorithm; Test case; Monte Carlo method; Mathematics; Machine learning; Statistics","score_opus":0.04507286624197506,"score_gpt":0.2673688255707413,"score_spread":0.22229595932876625,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167505893","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070254733,0.00013601284,0.98901176,0.00014100742,0.000034327666,0.0000652448,0.000011941611,0.00016336209,0.003410873],"genre_scores_gemma":[0.22242913,0.00032170038,0.7710112,0.000232926,0.000060093946,0.00030420284,0.00006995154,0.00007519664,0.005495672],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99922323,0.0002562032,0.000033050266,0.00012110841,0.0003140115,0.00005250908],"domain_scores_gemma":[0.9990318,0.00051928643,0.000082034254,0.00012846611,0.00020346917,0.00003502719],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086497783,0.00045888533,0.00045046565,0.000806731,0.00032026897,0.00052895333,0.0013547945,0.00075804023,0.002092464],"category_scores_gemma":[0.0033846707,0.00022413667,0.000601048,0.00053212605,0.000740531,0.0007367472,0.0006840605,0.00078189786,0.0003037462],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008245389,0.00015014123,0.0014441392,0.00018352484,0.00009170402,0.0005674187,0.00017036463,0.4335971,0.026038285,0.18685094,0.0022590621,0.3485648],"study_design_scores_gemma":[0.000036768863,0.00016829975,0.00038011087,0.000023150058,0.000030799187,0.0005191602,0.00001778362,0.9609537,0.0029493533,0.026101803,0.008795981,0.000023029412],"about_ca_topic_score_codex":0.0007219404,"about_ca_topic_score_gemma":0.0008744556,"teacher_disagreement_score":0.002092464,"about_ca_system_score_codex":0.00038577957,"about_ca_system_score_gemma":0.00061007694,"threshold_uncertainty_score":0.0069999695},"labels":[],"label_agreement":null},{"id":"W2167680630","doi":"10.1145/2491404.2491406","title":"Efficiency of subtype test in object oriented languages with generics","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Programming language; Test (biology); Semantics (computer science); Typing; Object-oriented programming","score_opus":0.008265875011669218,"score_gpt":0.23454597693097187,"score_spread":0.22628010191930265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167680630","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7857477,0.0017636757,0.1964267,0.0009304648,0.0000759962,0.00017921066,0.00027317842,0.008127758,0.0064753764],"genre_scores_gemma":[0.9049137,0.00042174335,0.09148658,0.00020419902,0.00003841663,0.000057132733,0.00035587445,0.0013935473,0.0011288541],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9653016,0.017842812,0.0033637816,0.0020397422,0.0076118056,0.0038401396],"domain_scores_gemma":[0.817841,0.11882436,0.013172097,0.041553687,0.0068245567,0.0017842837],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022556724,0.0011594853,0.0021751928,0.0026270347,0.0014696723,0.003803567,0.004079554,0.0022743791,0.0024129108],"category_scores_gemma":[0.09590813,0.0011681069,0.0014486127,0.0030847904,0.003143078,0.008817825,0.0034323006,0.0017049181,0.00076967204],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009560425,0.0013421545,0.10671889,0.0012522066,0.00051614916,0.0011820965,0.0018456148,0.22284876,0.07378646,0.0685798,0.006664543,0.50570285],"study_design_scores_gemma":[0.0004511223,0.0011812885,0.011502691,0.00019229478,0.00053443754,0.001173591,0.0007074171,0.841563,0.097471,0.041099105,0.003946265,0.00017775662],"about_ca_topic_score_codex":0.006085016,"about_ca_topic_score_gemma":0.0059111724,"teacher_disagreement_score":0.022556724,"about_ca_system_score_codex":0.0025226453,"about_ca_system_score_gemma":0.0038624553,"threshold_uncertainty_score":0.119292796},"labels":[],"label_agreement":null},{"id":"W2167924718","doi":"10.1109/compsac.2008.123","title":"Mutation-Based Testing of Buffer Overflow Vulnerabilities","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Buffer overflow; Computer science; Vulnerability (computing); Secure coding; Fuzz testing; Vulnerability management; Set (abstract data type); Software; Vulnerability assessment; Computer security; Software security assurance; Programming language; Information security","score_opus":0.04601465034739637,"score_gpt":0.2578013310027019,"score_spread":0.21178668065530554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167924718","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6744246,0.0001964141,0.3197999,0.00023113668,0.00003365034,0.00014333501,0.000118705095,0.0031729273,0.0018793629],"genre_scores_gemma":[0.9326121,0.00007712139,0.06650775,0.00007680886,0.000007004328,0.00008348462,0.00013900994,0.00008114481,0.00041563043],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99851865,0.00049384235,0.000086594744,0.00021435718,0.0005665014,0.00012009013],"domain_scores_gemma":[0.99602365,0.0026801242,0.00044642846,0.00031277267,0.00042246142,0.00011455974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011687502,0.0006626447,0.0004024881,0.0013473211,0.00026970354,0.00042324862,0.0009499033,0.0007025633,0.00061584095],"category_scores_gemma":[0.0062433504,0.00014704147,0.000445687,0.00056629546,0.0008971646,0.00087755435,0.0008184817,0.00050266815,0.000083126455],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072450936,0.00065282965,0.031625215,0.00039669068,0.00018135585,0.0028869512,0.00096798735,0.2531591,0.37826425,0.035341714,0.002222102,0.29357728],"study_design_scores_gemma":[0.00010076133,0.00063417846,0.006301233,0.000051515755,0.00007983148,0.0011863786,0.000099141,0.8016902,0.17425081,0.013428547,0.0021139323,0.00006357356],"about_ca_topic_score_codex":0.0009931282,"about_ca_topic_score_gemma":0.00078999094,"teacher_disagreement_score":0.0013473211,"about_ca_system_score_codex":0.00042920644,"about_ca_system_score_gemma":0.00055189704,"threshold_uncertainty_score":0.0061810017},"labels":[],"label_agreement":null},{"id":"W2169207812","doi":"10.1109/qsic.2006.46","title":"Optimal Synchronizable Test Sequence from Test Segments","year":2006,"lang":"en","type":"article","venue":"Proceedings of the ... International Conference on Quality Software/Proceedings","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Controllability; Computer science; Sequence (biology); Test (biology); Conformance testing; Context (archaeology); System under test; Test case; Machine learning; Mathematics","score_opus":0.05284735757345193,"score_gpt":0.3021675926522371,"score_spread":0.24932023507878515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169207812","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14252295,0.00023023992,0.85077566,0.00018946028,0.00006319701,0.00042259105,0.00015318298,0.001820192,0.0038225797],"genre_scores_gemma":[0.7642819,0.00011628747,0.2324863,0.00009776347,0.00004117631,0.0005382532,0.00064749847,0.00022462761,0.0015660471],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982395,0.0006168074,0.00012568316,0.00038949517,0.00040046877,0.00022813345],"domain_scores_gemma":[0.99647766,0.0015963952,0.00053118734,0.00051150133,0.0006115013,0.00027168432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011089293,0.0011840746,0.001134061,0.0012116659,0.00046502266,0.0006479155,0.0009212607,0.00074596034,0.0031319181],"category_scores_gemma":[0.0060081896,0.00046970576,0.00051227305,0.00080172444,0.00077332073,0.00094750576,0.0009910138,0.0009214697,0.00041175055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002783756,0.00054434227,0.0023522081,0.0004622307,0.0000695928,0.00083482615,0.0004042073,0.5149031,0.08144272,0.042418156,0.0026233396,0.35116145],"study_design_scores_gemma":[0.0004868021,0.0014937846,0.0019217697,0.00007956888,0.00008375617,0.00038968312,0.00019202278,0.8940594,0.046697382,0.050100684,0.0044527473,0.000042402105],"about_ca_topic_score_codex":0.00080049044,"about_ca_topic_score_gemma":0.0009284772,"teacher_disagreement_score":0.0031319181,"about_ca_system_score_codex":0.0006776759,"about_ca_system_score_gemma":0.0018733356,"threshold_uncertainty_score":0.010477364},"labels":[],"label_agreement":null},{"id":"W2169461503","doi":"10.1109/apsec.1994.465263","title":"Automated class testing: methods and experience","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Oracle; Computer science; Class (philosophy); Object (grammar); Contrast (vision); Programming language; Test (biology); Theoretical computer science; Artificial intelligence","score_opus":0.094927932525324,"score_gpt":0.3667322613879026,"score_spread":0.2718043288625786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169461503","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027968641,0.0035095576,0.9763534,0.0010124992,0.000097026226,0.00015844649,0.00010784881,0.0045537315,0.011410634],"genre_scores_gemma":[0.05428372,0.004993604,0.92499614,0.00039793848,0.00023872106,0.00033265748,0.0006275452,0.0017863531,0.0123432465],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9911322,0.0037155403,0.00050352124,0.0012752806,0.0030807583,0.00029270115],"domain_scores_gemma":[0.97931457,0.012623585,0.00052795076,0.004742303,0.002128106,0.0006634536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008504795,0.0016522778,0.0011024541,0.0028912316,0.0008473105,0.0049516005,0.0049533066,0.0016986746,0.015217363],"category_scores_gemma":[0.024784131,0.0014473784,0.001048803,0.0025486373,0.003905632,0.008001412,0.0029081728,0.004120946,0.006414124],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000115660616,0.000288099,0.0018715699,0.00039458368,0.000033189575,0.00013360176,0.00085223006,0.008041565,0.0022770714,0.12873301,0.019783508,0.8374759],"study_design_scores_gemma":[0.0002206655,0.00026468677,0.001932738,0.00089181203,0.000051358085,0.0017052904,0.0005577342,0.14892799,0.010560138,0.42707407,0.40762496,0.00018855449],"about_ca_topic_score_codex":0.002813936,"about_ca_topic_score_gemma":0.0019417105,"teacher_disagreement_score":0.015217363,"about_ca_system_score_codex":0.0019006848,"about_ca_system_score_gemma":0.0017018556,"threshold_uncertainty_score":0.050907135},"labels":[],"label_agreement":null},{"id":"W2170327261","doi":"10.1109/ccece.1997.614838","title":"The event-flow technique for selecting test cases for object-oriented programs","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Event (particle physics); Test (biology); Flow (mathematics); Mathematics","score_opus":0.03880748932803603,"score_gpt":0.2857414402070478,"score_spread":0.24693395087901177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170327261","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002330025,0.00005325266,0.99529713,0.000050348073,0.000016302192,0.00014028362,0.000037421683,0.0014033976,0.00067188405],"genre_scores_gemma":[0.092145756,0.0001628138,0.90494055,0.00018214292,0.0000547825,0.00045231427,0.00029550196,0.0003424568,0.0014237043],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99561834,0.0017258485,0.00027312723,0.000380161,0.0017523582,0.00025019314],"domain_scores_gemma":[0.98916036,0.00810009,0.00056220085,0.0010767292,0.00092136266,0.00017918176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044282773,0.0015477579,0.00068768073,0.0027971484,0.0006446845,0.0010072744,0.0018641585,0.0013092895,0.0039645573],"category_scores_gemma":[0.016899973,0.0007018627,0.0010898198,0.0011897002,0.0015253971,0.0027453653,0.0010670034,0.0020414684,0.0009159318],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007626003,0.00072231225,0.004486824,0.0007798016,0.00014305038,0.0011121549,0.0006567142,0.03688053,0.040572144,0.10275598,0.010093178,0.8010347],"study_design_scores_gemma":[0.0005781593,0.0017084646,0.0026403528,0.00041496215,0.00035221945,0.0033599644,0.0001717283,0.556173,0.1860138,0.19348395,0.054836407,0.00026691996],"about_ca_topic_score_codex":0.0011194258,"about_ca_topic_score_gemma":0.0011997456,"teacher_disagreement_score":0.0044282773,"about_ca_system_score_codex":0.00042593086,"about_ca_system_score_gemma":0.00085291,"threshold_uncertainty_score":0.023419261},"labels":[],"label_agreement":null},{"id":"W2170644445","doi":"10.1109/wcre.2008.27","title":"Data Model Reverse Engineering in Migrating a Legacy System to Java","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Java; Programming language; Programmer; Legacy system; Reverse engineering; Coding (social sciences); Software engineering; Data modeling; Software","score_opus":0.07230785660745945,"score_gpt":0.2711514458224755,"score_spread":0.19884358921501608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170644445","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39533037,0.00027506897,0.5888722,0.002756612,0.00017708474,0.0002654595,0.00016332041,0.0056921444,0.0064677107],"genre_scores_gemma":[0.50888324,0.00024587233,0.48339796,0.0007401339,0.00002423093,0.00013028912,0.00029437378,0.0014878066,0.00479616],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9916686,0.003583842,0.0006085542,0.0006587644,0.002954622,0.00052558305],"domain_scores_gemma":[0.9714641,0.012250948,0.0020247423,0.009584992,0.0043400587,0.00033520552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0111636305,0.00051559974,0.0003719876,0.0008793263,0.0016027396,0.0028562911,0.0021749432,0.001582268,0.0010480019],"category_scores_gemma":[0.04143218,0.00093282486,0.0006938722,0.00129651,0.0021825265,0.005683147,0.003140263,0.0039248774,0.00037129465],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008451791,0.0011826926,0.053431038,0.0008164031,0.00014118757,0.006449405,0.051113207,0.053772908,0.076534,0.14381452,0.010809176,0.60109025],"study_design_scores_gemma":[0.00024439636,0.00092442124,0.010557068,0.00058557285,0.00027579084,0.0066269445,0.0145794805,0.48162058,0.2660033,0.10117912,0.11703655,0.0003668227],"about_ca_topic_score_codex":0.0051857117,"about_ca_topic_score_gemma":0.011253677,"teacher_disagreement_score":0.0111636305,"about_ca_system_score_codex":0.0013516537,"about_ca_system_score_gemma":0.003240285,"threshold_uncertainty_score":0.059039652},"labels":[],"label_agreement":null},{"id":"W2171074827","doi":"10.1145/2687148.2687151","title":"Verifying security patches","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Correctness; Credence; Sketch; Semantics (computer science); Set (abstract data type); Program analysis; Computer security; Nothing; Programming language; Theoretical computer science; Algorithm; Machine learning","score_opus":0.011135649131328113,"score_gpt":0.2292289484173727,"score_spread":0.2180932992860446,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171074827","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026420586,0.00013888183,0.96797365,0.00055084843,0.00009402019,0.00014091858,0.00011108904,0.002600447,0.001969584],"genre_scores_gemma":[0.55165523,0.00029776467,0.44290906,0.0005499104,0.00018366019,0.00028678507,0.0004631872,0.0013139391,0.0023404236],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98540616,0.0033628996,0.0014365487,0.0027811064,0.00565436,0.0013588372],"domain_scores_gemma":[0.94416803,0.028382972,0.003608385,0.016836688,0.006280619,0.00072334066],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011318177,0.001056325,0.0013576237,0.0017245756,0.0020279265,0.0038663577,0.0031073669,0.0029057153,0.0041317623],"category_scores_gemma":[0.06477905,0.0015455554,0.0028279044,0.0007004721,0.008761057,0.010996819,0.0053155445,0.0048161866,0.0009943197],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035434437,0.00017364217,0.012280696,0.00079999317,0.00016368258,0.0013658002,0.0025321776,0.060453895,0.03243043,0.7938858,0.0044668466,0.09109261],"study_design_scores_gemma":[0.00021348112,0.00042907274,0.0020791714,0.00031996844,0.00018539441,0.0011754682,0.00052377896,0.17947172,0.06368713,0.7242545,0.027504845,0.0001554083],"about_ca_topic_score_codex":0.002378669,"about_ca_topic_score_gemma":0.0012305101,"teacher_disagreement_score":0.011318177,"about_ca_system_score_codex":0.0018584701,"about_ca_system_score_gemma":0.0030358473,"threshold_uncertainty_score":0.05985695},"labels":[],"label_agreement":null},{"id":"W2171131535","doi":"10.1109/step.2005.30","title":"TETE: A Non-Invasive Unit Testing Framework for Source Transformation","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Unit testing; Computer science; Eclipse; Programming language; Transformation (genetics); Debugging; Model transformation; Software engineering; Simple (philosophy); Test-driven development; Test (biology); Program transformation; Grammar; Software development; Artificial intelligence; Software","score_opus":0.0467537272594929,"score_gpt":0.2946937454766724,"score_spread":0.24794001821717948,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171131535","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007075756,0.00005320682,0.9919567,0.00009277062,0.000022878687,0.000091133275,0.00006818371,0.006039835,0.00096757006],"genre_scores_gemma":[0.058537457,0.0002778984,0.93499994,0.00019797852,0.00005136644,0.00041596484,0.00055649085,0.0027029375,0.0022598582],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.994158,0.0019274725,0.0005367426,0.0005184759,0.0023889728,0.00047032023],"domain_scores_gemma":[0.99221957,0.0041269623,0.0006208022,0.0017699224,0.0009895744,0.00027320575],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006492694,0.0012463791,0.0010711618,0.0020290483,0.00077507924,0.003187149,0.003541354,0.0018339683,0.004043944],"category_scores_gemma":[0.015660856,0.0009680151,0.0018577512,0.0008758044,0.0031674632,0.0036122478,0.0030224838,0.0036966691,0.0014126867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002324334,0.0003403449,0.0028265705,0.00079948193,0.00013594645,0.001296543,0.0012997171,0.05440831,0.021239504,0.55651027,0.018017618,0.3428932],"study_design_scores_gemma":[0.00019106643,0.0003771982,0.0012710963,0.0006429441,0.00013060788,0.002490927,0.0002659478,0.39186108,0.0557282,0.37864196,0.16816705,0.00023184571],"about_ca_topic_score_codex":0.0032403371,"about_ca_topic_score_gemma":0.0034850086,"teacher_disagreement_score":0.006492694,"about_ca_system_score_codex":0.0010337061,"about_ca_system_score_gemma":0.0027665603,"threshold_uncertainty_score":0.034337044},"labels":[],"label_agreement":null},{"id":"W2171274571","doi":"10.1109/csmr.2005.10","title":"An XML-Based Framework for Language Neutral Program Representation and Generic Analysis","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Programming language; XML; Source code; Program comprehension; XML validation; XML framework; Document Structure Description; Efficient XML Interchange; Representation (politics); XML Schema Editor; Streaming XML; Program analysis; Static program analysis; Software development; Software; World Wide Web; Software system","score_opus":0.030951523715071736,"score_gpt":0.3701001120251534,"score_spread":0.33914858831008166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171274571","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00036893727,0.00008355124,0.9955776,0.00015785052,0.000029688099,0.00008207926,0.00009891004,0.0024974355,0.0011040028],"genre_scores_gemma":[0.012622148,0.0003584478,0.98251444,0.00020710952,0.000076095705,0.00032951796,0.0007616718,0.000940994,0.00218956],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99577904,0.0015136782,0.0007552406,0.00055670005,0.0011256391,0.00026961774],"domain_scores_gemma":[0.9960264,0.0012301165,0.0004082679,0.0014012125,0.00074192905,0.0001921389],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067672064,0.00140229,0.0010539553,0.0033478641,0.0013595445,0.005584031,0.0042869905,0.0027928557,0.0045429925],"category_scores_gemma":[0.008791774,0.0010916746,0.0029288616,0.0036467472,0.0030338445,0.0069359276,0.0036982484,0.0047639185,0.002664812],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006463629,0.00007416661,0.00047071648,0.00037463722,0.000048337322,0.00046253385,0.0010227446,0.0071951416,0.0060469704,0.8724317,0.010049812,0.1017586],"study_design_scores_gemma":[0.000063297455,0.00014593896,0.0004597688,0.00051209977,0.00013779035,0.0016416067,0.00029765556,0.10616987,0.01579788,0.45557162,0.4190605,0.00014196242],"about_ca_topic_score_codex":0.0024366525,"about_ca_topic_score_gemma":0.0023843572,"teacher_disagreement_score":0.0067672064,"about_ca_system_score_codex":0.0014204829,"about_ca_system_score_gemma":0.0025067793,"threshold_uncertainty_score":0.035788894},"labels":[],"label_agreement":null},{"id":"W2171494278","doi":"","title":"Software testing by active learning for commercial games","year":2005,"lang":"en","type":"article","venue":"National Conference on Artificial Intelligence","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of Alberta","funders":"","keywords":"Computer science; Correctness; Video game; Context (archaeology); Software engineering; Software performance testing; Artificial intelligence; White-box testing; Software; Active learning (machine learning); Machine learning; Human–computer interaction; Software construction; Software system; Multimedia; Programming language","score_opus":0.16779529998615517,"score_gpt":0.36722164474622604,"score_spread":0.19942634476007087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171494278","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.069560885,0.00023961786,0.92325276,0.0004333349,0.000024496068,0.00015075789,0.0000625362,0.0034394504,0.0028361697],"genre_scores_gemma":[0.73717594,0.00008410562,0.26069412,0.0000995688,0.000020469342,0.00023916589,0.00012837522,0.0001944123,0.0013638859],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99694306,0.0016297989,0.00013420297,0.00035307062,0.0007654254,0.0001744574],"domain_scores_gemma":[0.98177195,0.01500123,0.00061553094,0.0014770876,0.0009026655,0.00023149715],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034433014,0.000928541,0.0007636772,0.0011181032,0.0005558078,0.0016583115,0.0020810112,0.0012184224,0.0026313388],"category_scores_gemma":[0.017923873,0.00053765136,0.0007072483,0.0006846237,0.0015543633,0.0026298172,0.0016478044,0.0015818132,0.00039092533],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054842234,0.00066317286,0.0049954094,0.00018953465,0.00008383315,0.00022246911,0.00039963864,0.6073303,0.008224253,0.036364518,0.002576827,0.33840165],"study_design_scores_gemma":[0.00001115024,0.000023999755,0.00011542194,0.000003344444,0.0000023701414,0.000012294965,0.0000077022005,0.9914929,0.0013815569,0.0066447547,0.00030124708,0.0000031176023],"about_ca_topic_score_codex":0.004341784,"about_ca_topic_score_gemma":0.0044340957,"teacher_disagreement_score":0.004341784,"about_ca_system_score_codex":0.001462862,"about_ca_system_score_gemma":0.0009453083,"threshold_uncertainty_score":0.018210113},"labels":[],"label_agreement":null},{"id":"W2174629870","doi":"","title":"Fault model-driven test derivation from finite state models: annotated bibliography","year":2001,"lang":"en","type":"article","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Test suite; Fault coverage; Fault model; Automaton; Stuck-at fault; Conformance testing; Finite-state machine; Fault (geology); Equivalence (formal languages); TRACE (psycholinguistics); Equivalence relation; Algorithm; Fault injection; Fault detection and isolation; Theoretical computer science; Test case; Programming language; Software; Mathematics; Discrete mathematics; Artificial intelligence; Standardization; Engineering","score_opus":0.029094367605719408,"score_gpt":0.2628401739598293,"score_spread":0.23374580635410988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2174629870","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0035219805,0.10985539,0.82894176,0.0015523895,0.002047403,0.00031705745,0.003179563,0.0040920395,0.046492428],"genre_scores_gemma":[0.12390659,0.17595991,0.6276676,0.0012783576,0.0012712941,0.0008472877,0.01899899,0.0025299427,0.047540028],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99926716,0.00010779601,0.00013079964,0.00013818596,0.00032657263,0.000029595458],"domain_scores_gemma":[0.996861,0.0017662906,0.000114519,0.0003314713,0.00090208044,0.000024634299],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008868517,0.0009016293,0.00087785005,0.004813422,0.00050802936,0.0013594131,0.0017355377,0.0013069857,0.014628284],"category_scores_gemma":[0.0056442553,0.00081326393,0.0010339642,0.006351214,0.00052461284,0.001964465,0.00063756877,0.0010590506,0.005031822],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007660805,0.00015041098,0.00044968448,0.008867049,0.00010739182,0.00069928425,0.00032784673,0.041860677,0.004247834,0.062016543,0.09224398,0.7889527],"study_design_scores_gemma":[0.00007098106,0.00014434347,0.0013022432,0.006154228,0.0003859028,0.0020292148,0.00014086414,0.06448158,0.020018956,0.28555614,0.61956275,0.00015286983],"about_ca_topic_score_codex":0.0032038991,"about_ca_topic_score_gemma":0.004131156,"teacher_disagreement_score":0.014628284,"about_ca_system_score_codex":0.000951188,"about_ca_system_score_gemma":0.0016203853,"threshold_uncertainty_score":0.048936486},"labels":[],"label_agreement":null},{"id":"W2182335636","doi":"10.4018/978-1-60960-747-0.ch002","title":"Exceptions for Dependability","year":2011,"lang":"en","type":"book-chapter","venue":"Advances in computer and electrical engineering book series","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Correctness; Dependability; Mathematical proof; Computer science; Masking (illustration); Rollback; Programming language; Theoretical computer science; Software engineering; Database transaction; Mathematics","score_opus":0.011504289668309088,"score_gpt":0.22498159509788404,"score_spread":0.21347730542957497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2182335636","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034444327,0.13673514,0.21031119,0.009701271,0.0052274424,0.00012786896,0.00035842997,0.0016881013,0.63240606],"genre_scores_gemma":[0.15132582,0.14950952,0.1374431,0.00709035,0.007361594,0.00048279,0.0011070105,0.0016876679,0.5439921],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99916375,0.00015761996,0.000065880646,0.0001774677,0.00037976613,0.000055649067],"domain_scores_gemma":[0.9985985,0.00078804395,0.00007477786,0.00024849837,0.00024070156,0.000049429626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071488315,0.0011894346,0.00051977305,0.0012763585,0.0011006325,0.0027961512,0.0010624299,0.0013746638,0.016729644],"category_scores_gemma":[0.0026182313,0.00042627103,0.0005789983,0.0015443851,0.0029356885,0.0052917297,0.0015528367,0.0038643656,0.0062591326],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000088569905,0.000012656961,0.000055804987,0.00023499594,0.000006370404,0.00007457163,0.0003220074,0.00091202586,0.0004682739,0.8732197,0.038225144,0.086459525],"study_design_scores_gemma":[0.0000043083196,0.000015016906,0.00009162303,0.00026751184,0.000005189377,0.00023105189,0.000057680645,0.00087721867,0.00026682642,0.49056265,0.5076106,0.000010305055],"about_ca_topic_score_codex":0.0012505738,"about_ca_topic_score_gemma":0.001185985,"teacher_disagreement_score":0.016729644,"about_ca_system_score_codex":0.0018172546,"about_ca_system_score_gemma":0.0009466065,"threshold_uncertainty_score":0.055966258},"labels":[],"label_agreement":null},{"id":"W2202321577","doi":"10.1007/978-3-642-28891-3_6","title":"Symbolic Execution of Communicating and Hierarchically Composed UML-RT State Machines","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Reachability; Unified Modeling Language; Modular design; Programming language; Reuse; Symbolic execution; Finite-state machine; State (computer science); Theoretical computer science; Software","score_opus":0.019748807224966396,"score_gpt":0.2626622974765419,"score_spread":0.24291349025157552,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2202321577","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08436523,0.00012423263,0.900416,0.00011693413,0.000044529417,0.00009582665,0.00017610668,0.0055092312,0.009151765],"genre_scores_gemma":[0.6703657,0.000118748714,0.32386798,0.000040450643,0.000015335843,0.0001284281,0.000374136,0.0005771901,0.004511993],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984835,0.0005375152,0.000078698395,0.000181077,0.00054227916,0.00017698549],"domain_scores_gemma":[0.9969785,0.002157666,0.0002064182,0.00035415072,0.00023715623,0.00006618634],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010410888,0.00073164597,0.000595897,0.00053205277,0.00075025536,0.0011694017,0.0014911905,0.000872284,0.006207793],"category_scores_gemma":[0.004890731,0.000661939,0.0009033577,0.00053223333,0.0016738239,0.0014002069,0.0014100114,0.001001004,0.00066919066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005132812,0.00014674384,0.0017576168,0.00046175992,0.00008621132,0.0008407213,0.0013105881,0.6778632,0.040446147,0.17833024,0.0018355275,0.09640798],"study_design_scores_gemma":[0.0000463154,0.00005503343,0.00024003666,0.000039808696,0.000035683333,0.00006104171,0.00005731424,0.9377686,0.02131949,0.03775548,0.0026029642,0.000018245888],"about_ca_topic_score_codex":0.004511107,"about_ca_topic_score_gemma":0.0065303403,"teacher_disagreement_score":0.006207793,"about_ca_system_score_codex":0.001027332,"about_ca_system_score_gemma":0.0013176543,"threshold_uncertainty_score":0.020767152},"labels":[],"label_agreement":null},{"id":"W2205665593","doi":"","title":"The case for system testing with swift hierarchical VM fork","year":2014,"lang":"en","type":"article","venue":"IEEE International Conference on Cloud Computing Technology and Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Fork (system call); Computer science; Test suite; Cloud computing; Operating system; Unit testing; Virtual machine; Software performance testing; White-box testing; Suite; Embedded system; Software; Test case; Software system; Software construction","score_opus":0.04916953759013115,"score_gpt":0.3087123122288789,"score_spread":0.25954277463874775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2205665593","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3034215,0.0021904937,0.55084413,0.06913484,0.0014430917,0.0007826874,0.00031256228,0.016804518,0.05506618],"genre_scores_gemma":[0.7620985,0.00025788092,0.22492011,0.0062160185,0.00021319222,0.0002709542,0.00016202792,0.0010397425,0.0048215757],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9883376,0.003922747,0.00046848253,0.0015575138,0.003693989,0.0020197714],"domain_scores_gemma":[0.96391296,0.017076666,0.002252258,0.012104989,0.002725488,0.0019275296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011637051,0.0008107028,0.00054371473,0.0006739072,0.001353236,0.002186085,0.0035506764,0.003359194,0.0043795514],"category_scores_gemma":[0.038710304,0.00062242115,0.00085814315,0.0005843226,0.0035449434,0.009703865,0.0023650494,0.004828104,0.0014460256],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020226724,0.0013467396,0.064419314,0.0010748629,0.00022848832,0.012084395,0.0029924023,0.051766157,0.07179608,0.23661616,0.083677806,0.47197488],"study_design_scores_gemma":[0.000523878,0.003173394,0.024786504,0.0009947536,0.00023396363,0.016086875,0.0027373198,0.38228136,0.061124206,0.31617197,0.19154884,0.00033692975],"about_ca_topic_score_codex":0.0027949694,"about_ca_topic_score_gemma":0.0043723574,"teacher_disagreement_score":0.011637051,"about_ca_system_score_codex":0.0011318102,"about_ca_system_score_gemma":0.002328982,"threshold_uncertainty_score":0.061543345},"labels":[],"label_agreement":null},{"id":"W2243603742","doi":"10.1109/ase.2015.23","title":"Synthesizing Web Element Locators (T)","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Mitacs; Intel Corporation","keywords":"JavaScript; Computer science; Document Object Model; Web application; Element (criminal law); Code (set theory); Web service; Web page; World Wide Web; Programming language; Set (abstract data type)","score_opus":0.04862106141560102,"score_gpt":0.26953980481618445,"score_spread":0.22091874340058343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2243603742","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061697584,0.00014191437,0.9138622,0.00015053466,0.0000717211,0.0002961687,0.00086566835,0.01600849,0.0069057164],"genre_scores_gemma":[0.2031503,0.00011148766,0.78873533,0.00010313912,0.000012960333,0.00029777514,0.0014533758,0.0029186574,0.0032170175],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99864954,0.00028431052,0.00013541356,0.0002955891,0.00054027775,0.00009500015],"domain_scores_gemma":[0.9937689,0.0033442008,0.0004195762,0.0009262064,0.0014387757,0.000102335456],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010458172,0.001026832,0.00048884074,0.0012995797,0.00040559177,0.0010539176,0.0010386118,0.0010760608,0.0047731516],"category_scores_gemma":[0.009243893,0.0006550839,0.00091067835,0.0005615398,0.00066081266,0.0013068931,0.0011562108,0.0007744074,0.0017719369],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042193336,0.0003634904,0.011338662,0.0020682546,0.000119313605,0.0023287954,0.0018824849,0.16390969,0.28238624,0.02958654,0.012506931,0.49308768],"study_design_scores_gemma":[0.000082317965,0.00021946256,0.0010559827,0.00015087893,0.00008730454,0.00079584954,0.00033402428,0.5786361,0.36396912,0.008562507,0.046043757,0.00006267822],"about_ca_topic_score_codex":0.0015151189,"about_ca_topic_score_gemma":0.0023137378,"teacher_disagreement_score":0.0047731516,"about_ca_system_score_codex":0.00055840565,"about_ca_system_score_gemma":0.0010501787,"threshold_uncertainty_score":0.015967727},"labels":[],"label_agreement":null},{"id":"W2246639849","doi":"10.1109/ase.2015.26","title":"Generating Fixtures for JavaScript Unit Testing (T)","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"JavaScript; Unobtrusive JavaScript; Computer science; Document Object Model; Unit testing; Test fixture; Fixture; Test case; Programming language; Web application; Rich Internet application; Operating system; Web page; Machine learning; World Wide Web; Software","score_opus":0.1726648636456591,"score_gpt":0.31919885295152883,"score_spread":0.14653398930586972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2246639849","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08357799,0.0002913166,0.8923897,0.00021205966,0.000038035327,0.00026892478,0.0005463744,0.020112377,0.002563256],"genre_scores_gemma":[0.4660436,0.00015728165,0.5285432,0.00018248349,0.000023837756,0.00034110906,0.0012560196,0.002476902,0.00097549864],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9966624,0.0011215444,0.0002648389,0.00058669056,0.0010921861,0.00027231913],"domain_scores_gemma":[0.9729106,0.019989863,0.0017742956,0.0034166018,0.0015779941,0.00033067487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024952877,0.0015274797,0.0007252144,0.0025333082,0.0006272649,0.0012043294,0.0023342152,0.0014696795,0.0039520254],"category_scores_gemma":[0.028155347,0.0006646579,0.001199321,0.0012562692,0.001751818,0.0016683224,0.0022458697,0.0011818694,0.00095717126],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006945003,0.00042375314,0.018341077,0.0012521205,0.00017945882,0.0018499944,0.0012615968,0.17521203,0.086296275,0.027462604,0.009588962,0.67743766],"study_design_scores_gemma":[0.00018235605,0.00064801663,0.0027578408,0.00023634908,0.0001383129,0.001504558,0.0002367484,0.74392974,0.2121936,0.027283678,0.010781573,0.000107260385],"about_ca_topic_score_codex":0.0023439073,"about_ca_topic_score_gemma":0.0020269435,"teacher_disagreement_score":0.0039520254,"about_ca_system_score_codex":0.0010185412,"about_ca_system_score_gemma":0.0014478699,"threshold_uncertainty_score":0.013220847},"labels":[],"label_agreement":null},{"id":"W2247591691","doi":"10.1109/ase.2015.102","title":"GRT: An Automated Test Generator Using Orchestrated Program Analysis","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Generator (circuit theory); Computer science; Test (biology); Automatic test equipment; Engineering; Reliability engineering; Power (physics)","score_opus":0.1083915520828959,"score_gpt":0.37862138673583295,"score_spread":0.27022983465293704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2247591691","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012047281,0.000101046106,0.7932495,0.00011893896,0.000058751724,0.0004769265,0.00084883446,0.19107704,0.0020217376],"genre_scores_gemma":[0.21572252,0.00023447814,0.75225717,0.0003305307,0.00006880502,0.001308582,0.006402691,0.020015344,0.0036598397],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9980823,0.00061082124,0.00015401766,0.00039705657,0.00059428194,0.00016156222],"domain_scores_gemma":[0.99534947,0.002422264,0.00039596733,0.001115911,0.0005292068,0.00018723804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019717042,0.0018593647,0.0006847623,0.0020429261,0.00027535943,0.0010295572,0.0021637825,0.000939519,0.007389792],"category_scores_gemma":[0.008886243,0.00065260596,0.0010045321,0.0009230276,0.0007831104,0.0013559484,0.0017089165,0.00096078345,0.0036856793],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011550548,0.00062314956,0.011265132,0.0010785126,0.00028988795,0.0030314662,0.00063531287,0.09648247,0.083612785,0.013343655,0.07774125,0.7107413],"study_design_scores_gemma":[0.00053066236,0.0005817862,0.0029073008,0.0001410366,0.000109362685,0.0023352206,0.00010690574,0.8554312,0.08508727,0.016690273,0.035952713,0.0001262197],"about_ca_topic_score_codex":0.0010343406,"about_ca_topic_score_gemma":0.00078353065,"teacher_disagreement_score":0.007389792,"about_ca_system_score_codex":0.00040582984,"about_ca_system_score_gemma":0.0010987907,"threshold_uncertainty_score":0.024721265},"labels":[],"label_agreement":null},{"id":"W2271850540","doi":"10.1145/2789209","title":"Test Case Prioritization Using Extended Digraphs","year":2015,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Regression testing; Test case; Test suite; Digraph; Model-based testing; Machine learning; Prioritization; Data mining; Test (biology); Hidden Markov model; Fault detection and isolation; Artificial intelligence; Reliability engineering; Software; Regression analysis; Software development","score_opus":0.15832191192600087,"score_gpt":0.3445707515609084,"score_spread":0.18624883963490751,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2271850540","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024109783,0.00019473242,0.9716684,0.0001505265,0.000026514499,0.00016350627,0.00014649019,0.002418541,0.0011213769],"genre_scores_gemma":[0.56642234,0.00024519206,0.4296098,0.00021810373,0.000030268966,0.0002893061,0.00058838516,0.00025007993,0.0023465068],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981041,0.0006027333,0.0001551574,0.0004425025,0.000504749,0.00019075916],"domain_scores_gemma":[0.9944049,0.003574082,0.00046366637,0.00070250366,0.00060257205,0.00025227596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001682438,0.0011896515,0.0009545083,0.0024335368,0.00040315322,0.001089879,0.0019617595,0.00074451126,0.0025961848],"category_scores_gemma":[0.008681951,0.0008039783,0.000906834,0.0011627383,0.0006861479,0.0019298356,0.0011544852,0.0012379215,0.00043215023],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026858205,0.00026205316,0.00421592,0.00025260128,0.00012118399,0.00041268955,0.00024088769,0.60398823,0.016096942,0.018997489,0.0021849908,0.35295844],"study_design_scores_gemma":[0.000024362475,0.00006844501,0.00036579158,0.000015638447,0.000021042746,0.000078842495,0.000015540381,0.9839827,0.00350517,0.010808996,0.0011003459,0.00001306053],"about_ca_topic_score_codex":0.011422841,"about_ca_topic_score_gemma":0.01477972,"teacher_disagreement_score":0.011422841,"about_ca_system_score_codex":0.0016072661,"about_ca_system_score_gemma":0.0020955577,"threshold_uncertainty_score":0.022712708},"labels":[],"label_agreement":null},{"id":"W2277501150","doi":"10.1109/issrew.2015.7392048","title":"Getting more in less: The power of single/error annotations in category partition","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Pairwise comparison; Partition (number theory); Computer science; Basis (linear algebra); Process (computing); Function (biology); Algorithm; Data mining; Theoretical computer science; Mathematics; Artificial intelligence; Programming language","score_opus":0.08382940141853011,"score_gpt":0.3080654015788923,"score_spread":0.22423600016036216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2277501150","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01656399,0.00030928076,0.97441566,0.0011450774,0.000099804165,0.00011157791,0.000118024655,0.0024473043,0.0047893873],"genre_scores_gemma":[0.32775465,0.0002581349,0.66637653,0.0006538406,0.00015041002,0.00024513624,0.00033903183,0.001426756,0.0027954539],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9709663,0.015235563,0.00095924997,0.003915562,0.0077393567,0.0011839499],"domain_scores_gemma":[0.83113253,0.12020646,0.0061777476,0.032022696,0.008622166,0.0018384894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020010514,0.0014698202,0.0015787526,0.0038372288,0.0034194477,0.0055775684,0.0058399467,0.00431817,0.006020054],"category_scores_gemma":[0.110938646,0.0014400629,0.0013986903,0.003594926,0.01586787,0.027644785,0.011174659,0.0060891616,0.0015376938],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010622187,0.00023995282,0.006484768,0.0003662813,0.00005718707,0.0005662632,0.005331827,0.033591896,0.0055434955,0.55375314,0.007642845,0.38536018],"study_design_scores_gemma":[0.00007868341,0.00022191953,0.00096413447,0.00025024218,0.00008156087,0.0004682252,0.000702235,0.17254163,0.009841263,0.79737973,0.017323107,0.00014725422],"about_ca_topic_score_codex":0.0065323575,"about_ca_topic_score_gemma":0.006604431,"teacher_disagreement_score":0.020010514,"about_ca_system_score_codex":0.0021948833,"about_ca_system_score_gemma":0.0032950507,"threshold_uncertainty_score":0.105826974},"labels":[],"label_agreement":null},{"id":"W2279957488","doi":"10.1109/issrew.2015.7392039","title":"An analysis and extension of Category partition testing for constrained systems","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Software testing; Software performance testing; System testing; Integration testing; White-box testing; Test strategy; Non-regression testing; Extension (predicate logic); Model-based testing; Software reliability testing; Task (project management); Keyword-driven testing; Partition (number theory); System integration testing; Software; Software system; Software engineering; Test case; Programming language; Machine learning; Software construction; Mathematics; Engineering; Systems engineering","score_opus":0.07935491540892278,"score_gpt":0.30696917360457493,"score_spread":0.22761425819565215,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2279957488","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05581439,0.00059062487,0.92744875,0.0006008592,0.00011112519,0.00015318335,0.00018930364,0.00040271893,0.014689062],"genre_scores_gemma":[0.8343348,0.0005755841,0.15603212,0.0004524814,0.00042035116,0.00017354915,0.0004323906,0.00029699635,0.0072817914],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9970396,0.00065946335,0.00008241535,0.00045847762,0.001265073,0.0004949615],"domain_scores_gemma":[0.9825812,0.012975774,0.0007718422,0.0013111854,0.0019213735,0.00043877054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023046078,0.0009495595,0.0009661433,0.003211347,0.0008148522,0.0014321976,0.0025386324,0.001256201,0.0053840545],"category_scores_gemma":[0.021348823,0.000518935,0.0020811963,0.0024301764,0.0034749182,0.003804617,0.0025916782,0.002186295,0.0004037453],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002643897,0.00020313516,0.0053782538,0.0004177556,0.00012158563,0.0008852391,0.000683497,0.27093977,0.0059287874,0.60052633,0.0048894873,0.10976166],"study_design_scores_gemma":[0.000014995214,0.00010192681,0.0016736342,0.000076563556,0.00005852885,0.00033615856,0.000054321838,0.61031675,0.0014171881,0.38285655,0.0030555306,0.000037783982],"about_ca_topic_score_codex":0.0056546056,"about_ca_topic_score_gemma":0.0034743107,"teacher_disagreement_score":0.0056546056,"about_ca_system_score_codex":0.0014929951,"about_ca_system_score_gemma":0.0017003752,"threshold_uncertainty_score":0.01801145},"labels":[],"label_agreement":null},{"id":"W2284923418","doi":"10.1007/978-3-642-38171-3_29","title":"Constraint-Based Fitness Function for Search-Based Software Testing","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Arity; Computer science; Fitness function; Software; Statement (logic); Constraint (computer-aided design); Function (biology); Theoretical computer science; Programming language; Machine learning; Mathematics","score_opus":0.044692900339624184,"score_gpt":0.26765899978498825,"score_spread":0.22296609944536405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2284923418","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014175922,0.00043244526,0.98067975,0.000113904,0.000045927904,0.000054285993,0.0000927379,0.00042703762,0.003977977],"genre_scores_gemma":[0.49312153,0.00039835428,0.4992051,0.000122016725,0.00006213026,0.00028475723,0.00047748705,0.00037994416,0.0059486683],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989992,0.00032864272,0.000056073994,0.000099325065,0.00043713994,0.0000795524],"domain_scores_gemma":[0.99805695,0.0013062891,0.00009060083,0.00013332022,0.0003678438,0.000044985172],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012418134,0.0009015863,0.0010331421,0.0011946908,0.00036768892,0.00078593806,0.0016159465,0.0013157562,0.0043467716],"category_scores_gemma":[0.0060963547,0.0002850977,0.0007303488,0.0014004243,0.00057118473,0.0009583621,0.0008086998,0.0010356831,0.00047150344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013609197,0.000120780955,0.00083390885,0.00015757387,0.000064003594,0.00013136705,0.000057536443,0.7281257,0.0076104756,0.026480455,0.0033964317,0.23288561],"study_design_scores_gemma":[0.000012750952,0.000028915843,0.00020556261,0.00001431292,0.000011639253,0.000041496896,0.000004078342,0.9928243,0.00073039875,0.0055106264,0.0006090979,0.0000068080094],"about_ca_topic_score_codex":0.0050160782,"about_ca_topic_score_gemma":0.0034647814,"teacher_disagreement_score":0.0050160782,"about_ca_system_score_codex":0.0008162549,"about_ca_system_score_gemma":0.0008343335,"threshold_uncertainty_score":0.014541328},"labels":[],"label_agreement":null},{"id":"W2294018077","doi":"10.1109/hase.2016.45","title":"An Extension of Category Partition Testing for Highly Constrained Systems","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Extension (predicate logic); Partition (number theory); White-box testing; Set (abstract data type); Base (topology); Black-box testing; Mathematical optimization; Software; Theoretical computer science; Algorithm; Programming language; Mathematics; Software system; Software construction","score_opus":0.0531951948667412,"score_gpt":0.284574951903722,"score_spread":0.2313797570369808,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2294018077","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040013414,0.00020386785,0.948203,0.00023403074,0.00006561229,0.00025569292,0.00010471683,0.00051792024,0.010401802],"genre_scores_gemma":[0.57382846,0.00020818332,0.42153987,0.00026969227,0.00006884469,0.00034476916,0.00025620018,0.00019812764,0.0032858676],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9948007,0.0014341442,0.0001721714,0.00058745116,0.0026064422,0.00039907833],"domain_scores_gemma":[0.9914185,0.0050033326,0.000592866,0.0014262794,0.0013138398,0.00024520294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019703347,0.000828003,0.0007520936,0.0018293312,0.00066468865,0.0014194262,0.0020609843,0.0013044551,0.0030198654],"category_scores_gemma":[0.010874857,0.0003974506,0.0010704268,0.001837347,0.0027792896,0.002351494,0.0025373877,0.0014259476,0.00042226652],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004716744,0.00035539447,0.006289403,0.00053102965,0.00011325459,0.0016566806,0.0012636344,0.23058958,0.04080405,0.42721003,0.004001602,0.2867137],"study_design_scores_gemma":[0.000070758484,0.000772276,0.003596776,0.00014806975,0.00005852462,0.00141546,0.00016609363,0.6486345,0.021789877,0.30000052,0.02320767,0.00013946938],"about_ca_topic_score_codex":0.0029818153,"about_ca_topic_score_gemma":0.002849037,"teacher_disagreement_score":0.0030198654,"about_ca_system_score_codex":0.00090596126,"about_ca_system_score_gemma":0.001526683,"threshold_uncertainty_score":0.010420263},"labels":[],"label_agreement":null},{"id":"W2296668858","doi":"10.1007/978-3-319-25945-1_2","title":"Using Multiple Adaptive Distinguishing Sequences for Checking Sequence Generation","year":2015,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Sequence (biology); Identification (biology); State (computer science); Computer science; Construct (python library); Model checking; Finite-state machine; Algorithm; Programming language","score_opus":0.2198040413080693,"score_gpt":0.3446613266231766,"score_spread":0.12485728531510731,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2296668858","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027571132,0.00014352244,0.96365523,0.00006086606,0.000072821356,0.00010533594,0.00010449409,0.0071919966,0.001094647],"genre_scores_gemma":[0.33499283,0.00006149152,0.6620518,0.00012686933,0.00003502943,0.00012595564,0.00032320071,0.00087459147,0.0014083216],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99428904,0.0015354013,0.0005295579,0.0015349649,0.0016852329,0.00042575118],"domain_scores_gemma":[0.97389245,0.014658666,0.0015341471,0.0065972055,0.0028705134,0.0004470212],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003183004,0.0013206243,0.0011018824,0.0029204683,0.00075924647,0.0012162472,0.003043801,0.0014128636,0.004317678],"category_scores_gemma":[0.016811432,0.00090748916,0.0010656117,0.0014723248,0.0015650004,0.0036761782,0.0027003468,0.0019797233,0.0011886214],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020208247,0.00037901505,0.0090387,0.0005582718,0.00016181583,0.0005681578,0.0005023552,0.052182786,0.090610236,0.045653187,0.003824107,0.7945005],"study_design_scores_gemma":[0.000174282,0.00055548095,0.0016638822,0.00015798735,0.00017938284,0.0007635788,0.000090895104,0.76760375,0.1577051,0.06290137,0.008088905,0.00011540911],"about_ca_topic_score_codex":0.0011661873,"about_ca_topic_score_gemma":0.0021591585,"teacher_disagreement_score":0.004317678,"about_ca_system_score_codex":0.00087960286,"about_ca_system_score_gemma":0.0016499747,"threshold_uncertainty_score":0.016833544},"labels":[],"label_agreement":null},{"id":"W2300062158","doi":"10.11575/prism/30915","title":"An Exploratory Study of Automated GUI Testing: Goals, Issues, and Best Practices","year":2014,"lang":"en","type":"article","venue":"Open MIND","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Test suite; Suite; Computer science; Graphical user interface testing; Software engineering; Keyword-driven testing; Best practice; Test (biology); Test strategy; Graphical user interface; Human–computer interaction; Test case; Programming language; User experience design; Machine learning; Software; Software development; Software construction","score_opus":0.15756735840381086,"score_gpt":0.4042491749741715,"score_spread":0.24668181657036065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2300062158","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9833681,0.00014293421,0.011642989,0.00086408516,0.000009580407,0.00039478953,0.000027026244,0.000056351495,0.003494024],"genre_scores_gemma":[0.97792155,0.0001461003,0.020727592,0.00021617196,0.000007748573,0.00032691273,0.00003229215,0.000030861087,0.00059086236],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9586857,0.03296549,0.0012756279,0.0014677893,0.0042309742,0.0013743679],"domain_scores_gemma":[0.72528565,0.23778208,0.010194758,0.0069052964,0.0152611965,0.0045710285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.042011615,0.0005061342,0.0005252508,0.0031310574,0.003385067,0.0045647398,0.0019554268,0.0018248941,0.0010409887],"category_scores_gemma":[0.1403653,0.0006599532,0.00035098055,0.0022857145,0.004798891,0.004550529,0.0031329826,0.0020115958,0.00019648665],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023686711,0.0024244208,0.107282944,0.00069492287,0.000041734555,0.0017721859,0.740594,0.0013444966,0.006589727,0.015194994,0.0017085086,0.12211505],"study_design_scores_gemma":[0.00027996814,0.0045735594,0.10999579,0.0013013651,0.000101045545,0.0028506725,0.7927747,0.021034762,0.0075306613,0.021974282,0.03734066,0.00024262161],"about_ca_topic_score_codex":0.0020901153,"about_ca_topic_score_gemma":0.0043685623,"teacher_disagreement_score":0.042011615,"about_ca_system_score_codex":0.0037125754,"about_ca_system_score_gemma":0.0045107175,"threshold_uncertainty_score":0.22218132},"labels":[],"label_agreement":null},{"id":"W2316535507","doi":"10.2316/p.2010.676-055","title":"Is Iterative Deepening Search a Complete and Optimal Algorithm for Java PathFinder?","year":2010,"lang":"en","type":"article","venue":"Parallel and Distributed Computing and Networks","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Pathfinder; Computer science; Java; Algorithm; Iterative method; Parallel computing; Operating system; World Wide Web","score_opus":0.024177024768679133,"score_gpt":0.2810436833405228,"score_spread":0.2568666585718437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2316535507","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01669259,0.00023320406,0.97705245,0.00046042446,0.000050785966,0.000047898695,0.00010824414,0.0030273644,0.0023270699],"genre_scores_gemma":[0.1471528,0.00013724061,0.848783,0.0002252415,0.00003872718,0.00010967466,0.00032826563,0.00081896636,0.0024061338],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980755,0.0005417669,0.00013016809,0.00050641893,0.00044798126,0.00029822555],"domain_scores_gemma":[0.9963547,0.001730038,0.00020456543,0.0011277945,0.0004768047,0.0001061728],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019641276,0.0008246023,0.0014911316,0.0011277082,0.00086541596,0.0015759374,0.0028219393,0.0017780311,0.007878045],"category_scores_gemma":[0.011653731,0.00070574216,0.0010525009,0.0012613528,0.0014049086,0.004860973,0.0020233218,0.001719572,0.001683978],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007326109,0.00021875027,0.0021176422,0.00031257886,0.00013643765,0.00007634694,0.00022568126,0.10022771,0.008054541,0.06434811,0.011173689,0.8123759],"study_design_scores_gemma":[0.00016756867,0.00013131232,0.00066094485,0.000058332822,0.00008478436,0.00016008157,0.00015252412,0.7948532,0.00939113,0.18645531,0.0078287795,0.000056059565],"about_ca_topic_score_codex":0.0039648833,"about_ca_topic_score_gemma":0.0059929146,"teacher_disagreement_score":0.007878045,"about_ca_system_score_codex":0.00080691674,"about_ca_system_score_gemma":0.002503057,"threshold_uncertainty_score":0.02635467},"labels":[],"label_agreement":null},{"id":"W2338921896","doi":"10.1109/tse.2016.2550441","title":"Test Case Prioritization Using Lexicographical Ordering","year":2016,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; University of Waterloo","keywords":"Lexicographical order; Computer science; Prioritization; Test case; Heuristic; Fault detection and isolation; Data mining; Fault (geology); Greedy algorithm; Reliability engineering; Algorithm; Machine learning; Artificial intelligence; Mathematics","score_opus":0.019273459654922546,"score_gpt":0.24197271955170754,"score_spread":0.222699259896785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2338921896","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03968177,0.00063365226,0.9466524,0.00041362012,0.00010983557,0.0011021936,0.00029924503,0.0023322226,0.008774976],"genre_scores_gemma":[0.21681851,0.00039904486,0.77866215,0.00024304162,0.00006595863,0.00040788465,0.0008610552,0.00035872156,0.002183746],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99458766,0.0017499481,0.0005448318,0.0005474344,0.0021919822,0.0003780957],"domain_scores_gemma":[0.98828596,0.006690451,0.00093275233,0.0013816061,0.0023577341,0.000351423],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021124894,0.0015072806,0.001344836,0.0060272943,0.001084929,0.0023874443,0.0016243877,0.00073615945,0.004009048],"category_scores_gemma":[0.01679645,0.00068780367,0.00094337517,0.0039748023,0.001042037,0.0015448299,0.0015018708,0.0014847069,0.0011699977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075218035,0.00069289905,0.005765572,0.0009322642,0.00016207338,0.00095831946,0.0008836312,0.105094776,0.050551347,0.038825557,0.008201285,0.78718007],"study_design_scores_gemma":[0.0004079413,0.0010146026,0.0036790725,0.00033087173,0.00025985035,0.0018881626,0.0007610202,0.80090374,0.06982267,0.09334195,0.027410148,0.0001800305],"about_ca_topic_score_codex":0.003542776,"about_ca_topic_score_gemma":0.006765392,"teacher_disagreement_score":0.0060272943,"about_ca_system_score_codex":0.0012816943,"about_ca_system_score_gemma":0.0037669935,"threshold_uncertainty_score":0.0134115815},"labels":[],"label_agreement":null},{"id":"W2339904014","doi":"10.1109/tse.2015.2487958","title":"Black-Box String Test Case Generation through a Multi-Objective Optimization","year":2015,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":53,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; String (physics); Random testing; Algorithm; Test case; Test (biology); Test functions for optimization; String searching algorithm; Optimization problem; Mathematics; Machine learning; Programming language; Data structure","score_opus":0.04811254978416045,"score_gpt":0.26542024025866834,"score_spread":0.21730769047450788,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2339904014","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046334747,0.0001853685,0.9501192,0.000117563046,0.000019938436,0.00026004566,0.00005651441,0.00078749115,0.002119124],"genre_scores_gemma":[0.43529698,0.00016028312,0.561384,0.00013520717,0.000015104285,0.0006177624,0.00021373376,0.00017864938,0.0019981724],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986695,0.0006117953,0.000066347675,0.00017954613,0.00037033335,0.000102472695],"domain_scores_gemma":[0.9963469,0.0026474055,0.00032877165,0.00016577494,0.00043211566,0.00007907952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019337004,0.0013275254,0.0009293341,0.001520473,0.00031509876,0.00068988174,0.001062233,0.001127921,0.002288584],"category_scores_gemma":[0.005252071,0.00049130537,0.000719195,0.0008271777,0.0006053451,0.00073930697,0.00097051874,0.00077963644,0.00034518115],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009628807,0.0001413607,0.0010317926,0.00012124499,0.00004255556,0.00019786842,0.00006351987,0.8965602,0.007137132,0.003969428,0.0006556612,0.08998288],"study_design_scores_gemma":[0.0000121624425,0.00007057523,0.00013110533,0.000008671258,0.000008645652,0.000029844785,0.0000074507084,0.99668866,0.0018181104,0.00095943746,0.0002611716,0.0000041338226],"about_ca_topic_score_codex":0.0012467526,"about_ca_topic_score_gemma":0.0010274397,"teacher_disagreement_score":0.002288584,"about_ca_system_score_codex":0.0005991488,"about_ca_system_score_gemma":0.0008723102,"threshold_uncertainty_score":0.010226488},"labels":[],"label_agreement":null},{"id":"W2343554492","doi":"","title":"Formal Approaches to Software Testing: Third International Workshop on Formal Approaches to Testing of Software: Fates 2003: Montreal, Quebec, Canada, October 6th, 2003: Revised Papers (LECTURE NOTES IN COMPUTER SCIENCE)","year":2004,"lang":"en","type":"book","venue":"Springer eBooks","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Software testing; Software engineering; Computer science; Formal methods; Library science; Software; Programming language","score_opus":0.0779619741570228,"score_gpt":0.23216008431352803,"score_spread":0.15419811015650522,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2343554492","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010295079,0.18037379,0.59077936,0.02143775,0.008165472,0.00053446746,0.0030419298,0.007480931,0.17789128],"genre_scores_gemma":[0.086303055,0.08970094,0.1673359,0.0017702361,0.0013790518,0.00041292797,0.005473277,0.003141901,0.64448273],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99892104,0.00015751451,0.00005368277,0.00012672081,0.0005788491,0.0001621973],"domain_scores_gemma":[0.9973253,0.00071619364,0.00008007819,0.00023150055,0.0014207532,0.00022615274],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023998031,0.0015272719,0.0013541412,0.002108644,0.0011870866,0.0043265917,0.0020579365,0.0011862371,0.03059329],"category_scores_gemma":[0.0027987568,0.0010456882,0.00079829694,0.003783216,0.0025522425,0.002408584,0.0008495346,0.0017453665,0.004848554],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009161774,0.00008827378,0.00053470465,0.0007570245,0.000029413748,0.00012393965,0.0009271168,0.0058049564,0.0039831544,0.0538186,0.43668944,0.49715185],"study_design_scores_gemma":[0.00008455293,0.00008124834,0.002913105,0.00083108887,0.00005522066,0.0003721734,0.0004094533,0.01543889,0.0052287006,0.042445134,0.93206084,0.00007959878],"about_ca_topic_score_codex":0.38864565,"about_ca_topic_score_gemma":0.4841393,"teacher_disagreement_score":0.38864565,"about_ca_system_score_codex":0.009744861,"about_ca_system_score_gemma":0.012774412,"threshold_uncertainty_score":0.77276695},"labels":[],"label_agreement":null},{"id":"W2363564475","doi":"","title":"Designing and implementing crawler-based Web test generation system","year":2014,"lang":"en","type":"article","venue":"Journal of Suzhou University of Science and Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Web crawler; Test (biology); Web testing; Automation; Web application; Test Management Approach; Computer science; Test harness; Test case; Software engineering; World Wide Web; Engineering; Web page; Web development; Web application security; Operating system; Machine learning; Software","score_opus":0.01438619461874731,"score_gpt":0.21097078333201505,"score_spread":0.19658458871326773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2363564475","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03884934,0.00024192632,0.8998257,0.00014930143,0.000064684675,0.00076668506,0.00018022349,0.05689972,0.0030224994],"genre_scores_gemma":[0.33256623,0.0002596883,0.658762,0.00031769084,0.00006116023,0.0009950532,0.0013383689,0.0017213691,0.0039783986],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977986,0.00052943855,0.00030923664,0.0004667331,0.0006971821,0.00019881324],"domain_scores_gemma":[0.9964135,0.0011853736,0.00031476867,0.0005172819,0.0013813073,0.00018773875],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017428485,0.0008004137,0.00092448684,0.0023157122,0.0005630264,0.0013150522,0.0017737657,0.0010450543,0.0027628108],"category_scores_gemma":[0.0049588233,0.0006423169,0.0005603587,0.00075528864,0.00039526098,0.001584956,0.00085124554,0.0009842564,0.0010989995],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008870286,0.0010052248,0.017672533,0.00098029,0.00026294112,0.0028497993,0.0011191021,0.040850166,0.18156481,0.010090311,0.019394396,0.7233234],"study_design_scores_gemma":[0.0005023581,0.00078776694,0.008770795,0.00017332824,0.0002507457,0.0026072748,0.0001722982,0.7375274,0.21143891,0.0030253832,0.03455896,0.0001848518],"about_ca_topic_score_codex":0.0025969208,"about_ca_topic_score_gemma":0.0016044367,"teacher_disagreement_score":0.0027628108,"about_ca_system_score_codex":0.00057630026,"about_ca_system_score_gemma":0.0013503599,"threshold_uncertainty_score":0.009242594},"labels":[],"label_agreement":null},{"id":"W2368534069","doi":"","title":"Functional Test Platform Research Based on Model Driven Technology","year":2009,"lang":"en","type":"article","venue":"Microcomputer applications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Test (biology)","score_opus":0.06005169548088268,"score_gpt":0.3248242107119926,"score_spread":0.2647725152311099,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2368534069","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028937163,0.00074524805,0.9622445,0.00039385416,0.0000772874,0.00012800953,0.000055727458,0.0010450768,0.0063730646],"genre_scores_gemma":[0.6095466,0.0017055478,0.38187662,0.0002611071,0.000099892844,0.0002905192,0.00030399178,0.00044501034,0.005470723],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99777406,0.0007363529,0.00006944068,0.00024417957,0.0010142031,0.0001617355],"domain_scores_gemma":[0.994833,0.0027963736,0.0002578993,0.0012101182,0.0007848362,0.0001177105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026818593,0.0012447158,0.00071156997,0.0017725402,0.00039356473,0.0016462408,0.002510059,0.0011385487,0.0033050885],"category_scores_gemma":[0.0077325557,0.0006665765,0.0009641189,0.00089103205,0.0014171334,0.004253262,0.0007194658,0.0018588719,0.00052643445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003711615,0.00060282694,0.003124316,0.0007387949,0.0001831577,0.00023327988,0.00029191622,0.15964669,0.06976289,0.42407247,0.0027351568,0.33823735],"study_design_scores_gemma":[0.0000887696,0.0012879536,0.0014810152,0.0001938801,0.00011654931,0.0004022986,0.00010760254,0.802239,0.060713988,0.11934024,0.013968863,0.000059865564],"about_ca_topic_score_codex":0.0014737374,"about_ca_topic_score_gemma":0.00085178355,"teacher_disagreement_score":0.0033050885,"about_ca_system_score_codex":0.0012515372,"about_ca_system_score_gemma":0.0014567779,"threshold_uncertainty_score":0.014183223},"labels":[],"label_agreement":null},{"id":"W2371137413","doi":"","title":"A Designing Method for Function Combination Testing Based on Orthogonal Chart","year":2007,"lang":"en","type":"article","venue":"Microcomputer applications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Chart; Test strategy; Software reliability testing; Software; Manual testing; Black-box testing; Keyword-driven testing; White-box testing; Reliability engineering; Test (biology); Function (biology); Non-regression testing; Software testing; Software engineering; Test Management Approach; Risk-based testing; Software construction; Software development; Programming language; Statistics; Mathematics","score_opus":0.032306658540416935,"score_gpt":0.3048874524493184,"score_spread":0.2725807939089015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2371137413","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019475729,0.00003173739,0.99541014,0.000040293908,0.000029412253,0.000099624755,0.000020936683,0.0010846675,0.0013357349],"genre_scores_gemma":[0.06857582,0.00012236123,0.9281336,0.000060749942,0.00004368407,0.00034102125,0.00013336066,0.00021585758,0.0023736174],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9962746,0.0010826804,0.0002687697,0.00070942554,0.001476909,0.00018763811],"domain_scores_gemma":[0.9966444,0.0014777699,0.00022506787,0.0005037478,0.0010435926,0.00010543359],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019398773,0.0011431598,0.000799909,0.002352895,0.0008060087,0.0011216432,0.0012169895,0.0007740094,0.0039652293],"category_scores_gemma":[0.006218952,0.00048465986,0.0009825736,0.001521431,0.0014248,0.0019186328,0.0010144723,0.0011234842,0.000829549],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002174043,0.00015427217,0.0024283768,0.00039618975,0.000061005267,0.000553656,0.0008891613,0.022324778,0.03217823,0.07552211,0.0075053717,0.85776937],"study_design_scores_gemma":[0.00036397204,0.0012935743,0.004119055,0.00025547016,0.0003071773,0.00466114,0.00039242627,0.7031886,0.092659734,0.07599183,0.116403654,0.0003634084],"about_ca_topic_score_codex":0.001913978,"about_ca_topic_score_gemma":0.0012877666,"teacher_disagreement_score":0.0039652293,"about_ca_system_score_codex":0.0006141122,"about_ca_system_score_gemma":0.0012154862,"threshold_uncertainty_score":0.013265014},"labels":[],"label_agreement":null},{"id":"W2382790792","doi":"","title":"Design of an Automatic Test Case Generator for Embedded C Compilers","year":2010,"lang":"en","type":"article","venue":"Microcomputer applications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Compiler; Programming language; Correctness; Generator (circuit theory); Semantics (computer science); Syntax; Algorithm; Artificial intelligence","score_opus":0.02220176430816151,"score_gpt":0.2852982312793528,"score_spread":0.26309646697119127,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2382790792","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010007654,0.00006495832,0.9749244,0.00008675688,0.000035051813,0.0005451058,0.00015184181,0.013452081,0.0007321292],"genre_scores_gemma":[0.2240575,0.00009590849,0.77038276,0.00018640318,0.000060231017,0.0016052562,0.0009559086,0.0014871835,0.0011688231],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99655557,0.0011842207,0.00035500914,0.00060371874,0.0010678604,0.00023361358],"domain_scores_gemma":[0.9912561,0.0050610634,0.00075042166,0.00097684,0.0017527353,0.00020268536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002538341,0.001185124,0.0009524816,0.0019942883,0.00031400958,0.0009450589,0.002498303,0.0009433091,0.0035453173],"category_scores_gemma":[0.009506641,0.0007213551,0.0006868201,0.00069283147,0.00060254533,0.0009567856,0.000720678,0.0010669616,0.0011878138],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012221311,0.00057764247,0.007834644,0.0010991205,0.0002647741,0.0020511688,0.0004918666,0.16359276,0.14569767,0.031310756,0.014654007,0.63120353],"study_design_scores_gemma":[0.00037561325,0.0005276785,0.0011015761,0.00007746591,0.00008298041,0.00125409,0.000036110057,0.8585048,0.1206298,0.0054771937,0.011858232,0.00007446109],"about_ca_topic_score_codex":0.0010033696,"about_ca_topic_score_gemma":0.00056078396,"teacher_disagreement_score":0.0035453173,"about_ca_system_score_codex":0.0006725909,"about_ca_system_score_gemma":0.0016019074,"threshold_uncertainty_score":0.013424158},"labels":[],"label_agreement":null},{"id":"W2389797271","doi":"","title":"State-Based Class Testing Research","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"CAE (Canada)","funders":"","keywords":"Computer science; System integration testing; Software reliability testing; Class (philosophy); Software performance testing; Non-regression testing; Software quality; Class diagram; State (computer science); Software engineering; Unit testing; Software construction; Software; Reliability engineering; Software development; Programming language; Unified Modeling Language; Engineering; Artificial intelligence","score_opus":0.24520556701404095,"score_gpt":0.354585413518932,"score_spread":0.10937984650489108,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2389797271","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029652983,0.009652391,0.90586597,0.0021884919,0.00060084625,0.00016581782,0.00016308071,0.0013015508,0.05040881],"genre_scores_gemma":[0.76132125,0.010369051,0.20676835,0.0009945866,0.0009904149,0.0003299587,0.0004966104,0.00034155205,0.018388277],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9955598,0.00097558385,0.00019536255,0.0010518886,0.0019659505,0.0002514219],"domain_scores_gemma":[0.9855773,0.008934893,0.00068746886,0.0022144977,0.0022844821,0.00030140285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028859426,0.00091788697,0.0008245261,0.0024379324,0.00070051313,0.0027370253,0.002407882,0.0013094192,0.0048426054],"category_scores_gemma":[0.013397279,0.0005094092,0.0011365722,0.0023906948,0.0031643286,0.0066584684,0.0012901876,0.0020612257,0.0009724235],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007173293,0.00015477047,0.002211327,0.0002581328,0.000057593144,0.000083328705,0.00033458776,0.023731539,0.0036403288,0.7074958,0.0031220864,0.25883883],"study_design_scores_gemma":[0.00007760025,0.0002821601,0.0018055043,0.00019817411,0.00010065459,0.00045650714,0.00012237829,0.36264473,0.011825612,0.56861347,0.05379187,0.00008133996],"about_ca_topic_score_codex":0.003987198,"about_ca_topic_score_gemma":0.0014941663,"teacher_disagreement_score":0.0048426054,"about_ca_system_score_codex":0.002707724,"about_ca_system_score_gemma":0.0014085687,"threshold_uncertainty_score":0.019645989},"labels":[],"label_agreement":null},{"id":"W2394617872","doi":"10.1016/b978-0-12-408094-2.00005-9","title":"Test Cost-Effectiveness and Defect Density: A Case Study on the Android Platform","year":2013,"lang":"en","type":"book-chapter","venue":"Advances in computers","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Orta Doğu Teknik Üniversitesi","keywords":"Android (operating system); Code coverage; Computer science; Replicate; Engineering; Embedded system; Operating system; Software; Statistics; Mathematics","score_opus":0.04311971506019773,"score_gpt":0.2986792452720685,"score_spread":0.2555595302118708,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2394617872","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99425393,0.00040584724,0.0019075322,0.00019385177,0.0000063174475,0.00007098486,0.00013596514,0.000033240565,0.0029924028],"genre_scores_gemma":[0.99651414,0.00020248856,0.002326137,0.000019893703,0.000005159215,0.000019287267,0.000064629916,0.000017057322,0.0008312968],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9982059,0.00073790766,0.00009112269,0.00015494492,0.0006269156,0.00018321368],"domain_scores_gemma":[0.9519196,0.043367315,0.0015810963,0.00075658225,0.001907683,0.00046771247],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022714508,0.0006096279,0.00043454906,0.0019062988,0.0006640716,0.0010857184,0.0014108764,0.0013253916,0.0020108484],"category_scores_gemma":[0.014314071,0.00027201037,0.0006347477,0.0020541297,0.0009153786,0.0011227545,0.0005842717,0.0007168367,0.00022603509],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002872322,0.008867952,0.51766485,0.0017334762,0.0005652454,0.040169746,0.010082943,0.10522456,0.018501164,0.010922013,0.006842105,0.2765536],"study_design_scores_gemma":[0.0004106933,0.0099043,0.5808626,0.00054128916,0.001281164,0.024327844,0.024970591,0.31701642,0.024728257,0.0086321365,0.007020418,0.0003043628],"about_ca_topic_score_codex":0.012174001,"about_ca_topic_score_gemma":0.018653918,"teacher_disagreement_score":0.012174001,"about_ca_system_score_codex":0.0015810783,"about_ca_system_score_gemma":0.0007264948,"threshold_uncertainty_score":0.02420628},"labels":[],"label_agreement":null},{"id":"W2397104696","doi":"10.1016/bs.adcom.2014.12.003","title":"Advances in Testing JavaScript-Based Web Applications","year":2015,"lang":"en","type":"book-chapter","venue":"Advances in computers","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Unobtrusive JavaScript; JavaScript; Computer science; Scripting language; Web application; Programming language; Oracle; Unit testing; Test script; Web testing; Software engineering; Rich Internet application; World Wide Web; Dynamic web page; Test case; Web development; Web service; Web application security; Software","score_opus":0.04193320955251735,"score_gpt":0.2981805410131842,"score_spread":0.25624733146066686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2397104696","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00775766,0.45159897,0.33512786,0.0031433525,0.0029834267,0.00012044486,0.0002949733,0.002630844,0.1963425],"genre_scores_gemma":[0.077620566,0.48884183,0.27227482,0.001747328,0.0031468088,0.00015699459,0.0013690288,0.0013007652,0.15354176],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99817884,0.00019103926,0.00012290366,0.00021013398,0.0012221931,0.000074950156],"domain_scores_gemma":[0.9965314,0.002138654,0.000102714235,0.00030111326,0.00082907657,0.00009704859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001290946,0.0009817001,0.0006893382,0.0028230129,0.00025388054,0.0016895055,0.0018147504,0.0009922349,0.007883637],"category_scores_gemma":[0.004423852,0.00053039804,0.00050793035,0.004670648,0.00074173557,0.0034501525,0.0012055994,0.0023523779,0.0035317154],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002071555,0.00009809855,0.00028278356,0.00082845724,0.000013059577,0.000079310455,0.00011218783,0.002766465,0.004714851,0.029532354,0.020073032,0.94147867],"study_design_scores_gemma":[0.00001725938,0.00018030935,0.0020932045,0.002207003,0.00007301362,0.0016624569,0.00011319426,0.023850525,0.020848699,0.09128869,0.85760766,0.000058009864],"about_ca_topic_score_codex":0.0010746998,"about_ca_topic_score_gemma":0.001480038,"teacher_disagreement_score":0.007883637,"about_ca_system_score_codex":0.0009123668,"about_ca_system_score_gemma":0.00090185355,"threshold_uncertainty_score":0.026373386},"labels":[],"label_agreement":null},{"id":"W2398007991","doi":"","title":"Machine Learning-based Software Testing: Towards a Classification Framework.","year":2011,"lang":"en","type":"article","venue":"Software Engineering and Knowledge Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University; University of New Brunswick","funders":"","keywords":"Computer science; Software testing; Software engineering; Software reliability testing; System integration testing; Software performance testing; Test strategy; Machine learning; Regression testing; Software construction; Artificial intelligence; Software; Software development; Programming language","score_opus":0.04277672093902538,"score_gpt":0.23928308148803581,"score_spread":0.19650636054901044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398007991","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03972584,0.007711474,0.93828547,0.003951135,0.00023172608,0.00038175852,0.0007248284,0.0018338762,0.007154018],"genre_scores_gemma":[0.6138121,0.0029762026,0.37544253,0.00082765264,0.00062506495,0.0004843348,0.0021429365,0.00016495971,0.003524217],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9939329,0.0021463572,0.000582054,0.0006685596,0.0023509767,0.00031913648],"domain_scores_gemma":[0.9701625,0.019017177,0.0025537848,0.002685161,0.0047891364,0.00079215725],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0066062966,0.0010376415,0.001177611,0.0059277187,0.00066046073,0.0030249245,0.0036055981,0.002315884,0.0016547091],"category_scores_gemma":[0.03479834,0.0002843371,0.0007895559,0.0041843317,0.0017515762,0.005228418,0.0014942205,0.0023226873,0.00083752285],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019050251,0.000669753,0.049634255,0.00068998896,0.00016153806,0.00027516313,0.00032708648,0.031875435,0.0021940668,0.06867548,0.013789459,0.83151734],"study_design_scores_gemma":[0.000050880742,0.00025692125,0.010850114,0.0004187011,0.00011161843,0.0007575756,0.00033065252,0.6946064,0.0063663656,0.27540055,0.010788048,0.00006221602],"about_ca_topic_score_codex":0.0021933913,"about_ca_topic_score_gemma":0.002532931,"teacher_disagreement_score":0.0066062966,"about_ca_system_score_codex":0.0012251682,"about_ca_system_score_gemma":0.0014780873,"threshold_uncertainty_score":0.03493786},"labels":[],"label_agreement":null},{"id":"W2398142207","doi":"","title":"Diagnosing new faults using mutants and prior faults.","year":2011,"lang":"en","type":"article","venue":"International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Reliability engineering; Engineering","score_opus":0.09920225159013185,"score_gpt":0.2892741667797259,"score_spread":0.19007191518959404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398142207","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51923543,0.0009916616,0.47097915,0.0003852454,0.00022638588,0.00017080053,0.00032858207,0.0058539263,0.0018289099],"genre_scores_gemma":[0.8581288,0.00014048815,0.14021574,0.00004632348,0.000019598141,0.000024957157,0.00036034177,0.00013933121,0.00092432735],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99821657,0.00034984862,0.00013502208,0.00045168094,0.0006915462,0.000155329],"domain_scores_gemma":[0.9870667,0.00701683,0.0012455594,0.0026158078,0.001653075,0.0004020989],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013252571,0.0010072427,0.00076202845,0.0018336603,0.00036339465,0.000999657,0.00179094,0.002014969,0.0011941987],"category_scores_gemma":[0.017586388,0.0004665476,0.0006252847,0.00048675924,0.0005218917,0.0026685314,0.00064068876,0.0011482342,0.0003445106],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001988663,0.0016922982,0.08734807,0.0006323741,0.00038846096,0.0035104535,0.00077067374,0.13985936,0.22622965,0.008061901,0.0032550015,0.5262631],"study_design_scores_gemma":[0.00012671998,0.0008842048,0.011746409,0.00009393807,0.00025439393,0.0021669166,0.0002010649,0.83825773,0.13394105,0.009605548,0.002636613,0.00008529861],"about_ca_topic_score_codex":0.0030974182,"about_ca_topic_score_gemma":0.004737412,"teacher_disagreement_score":0.0030974182,"about_ca_system_score_codex":0.0005715981,"about_ca_system_score_gemma":0.00082955643,"threshold_uncertainty_score":0.0070087314},"labels":[],"label_agreement":null},{"id":"W2401010683","doi":"","title":"Change-driven Incremental Symbolic Execution of Evolving State Machines.","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Symbolic execution; State (computer science); Programming language","score_opus":0.06153845280612907,"score_gpt":0.2896852166668992,"score_spread":0.22814676386077015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2401010683","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10201138,0.00029994408,0.88192326,0.00024356114,0.00008257702,0.00017688921,0.00027075628,0.009901166,0.0050905705],"genre_scores_gemma":[0.76654077,0.00010170054,0.23053531,0.00006362158,0.000015054345,0.00012220103,0.00046334116,0.00040548062,0.0017524933],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984524,0.0005201021,0.000058762067,0.00018438684,0.00063196325,0.0001522047],"domain_scores_gemma":[0.9923213,0.005098016,0.0003914647,0.0013473082,0.000681306,0.0001606954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012072145,0.0004895929,0.00046553914,0.00082942995,0.00039101462,0.0006678906,0.0017512334,0.00074047915,0.0034881865],"category_scores_gemma":[0.01349561,0.00037314018,0.00044937542,0.00076664484,0.0011481264,0.0016147186,0.0013778802,0.0011548805,0.00052945677],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011985451,0.00036995974,0.008265435,0.0005488079,0.0001385455,0.000897361,0.0010095082,0.43416506,0.046273254,0.07496961,0.0063822013,0.42578182],"study_design_scores_gemma":[0.000047134738,0.00009519808,0.00056750764,0.000022510572,0.00003269087,0.000095365365,0.000045898927,0.9583848,0.015810845,0.022977198,0.0019068499,0.000014023503],"about_ca_topic_score_codex":0.0033138602,"about_ca_topic_score_gemma":0.0052106334,"teacher_disagreement_score":0.0034881865,"about_ca_system_score_codex":0.00064428616,"about_ca_system_score_gemma":0.00085150474,"threshold_uncertainty_score":0.011669159},"labels":[],"label_agreement":null},{"id":"W2401568214","doi":"","title":"Combining static analysis and targeted symbolic execution for scalable bug-finding in application binaries","year":2016,"lang":"en","type":"article","venue":"Computer Science and Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Symbolic execution; Computer science; Concolic testing; Program analysis; Static analysis; Statement (logic); Programming language; Pruning; Process (computing); Program slicing; Scalability; Code (set theory); Debugging; Software; Operating system","score_opus":0.009698424595459712,"score_gpt":0.2327454978405086,"score_spread":0.22304707324504888,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2401568214","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11981735,0.0005780359,0.83438057,0.00028077865,0.000044582735,0.00019928225,0.00029743658,0.041718654,0.0026833883],"genre_scores_gemma":[0.63829637,0.0003400309,0.3570915,0.00013634113,0.000027459717,0.0001922784,0.0006971966,0.0018280824,0.0013906416],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983626,0.00041808267,0.00013135071,0.00030349984,0.0005704755,0.00021398546],"domain_scores_gemma":[0.99627906,0.0020329922,0.00039755212,0.0007943486,0.00036509483,0.00013098311],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001250889,0.0014604408,0.0008954309,0.002629257,0.0005078268,0.0010989095,0.0014502462,0.00064266165,0.0022126618],"category_scores_gemma":[0.0055145724,0.0007491583,0.0009681858,0.0016017914,0.0014254304,0.0021457819,0.0025956773,0.00090966816,0.00067356724],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008481812,0.00036615017,0.0127302855,0.0008725262,0.00019839144,0.00073463214,0.0007826631,0.2884261,0.08614955,0.0135128,0.0037576072,0.5916211],"study_design_scores_gemma":[0.000049271377,0.00016057387,0.0012096018,0.000055382294,0.00007762883,0.00013525247,0.00007466147,0.95349526,0.031698786,0.010649725,0.0023486589,0.000045255973],"about_ca_topic_score_codex":0.004554341,"about_ca_topic_score_gemma":0.00508123,"teacher_disagreement_score":0.004554341,"about_ca_system_score_codex":0.00083786726,"about_ca_system_score_gemma":0.0017926422,"threshold_uncertainty_score":0.009055674},"labels":[],"label_agreement":null},{"id":"W2401952677","doi":"","title":"Empirical Analysis for Investigating the Effect of Control Flow Dependencies on Testability of Classes.","year":2011,"lang":"en","type":"article","venue":"Software Engineering and Knowledge Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec","funders":"","keywords":"Testability; Computer science; Control flow; Data-flow analysis; Reliability engineering; Data flow diagram; Programming language; Engineering; Database","score_opus":0.025811940340620704,"score_gpt":0.25466925489641645,"score_spread":0.22885731455579575,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2401952677","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9755161,0.0009941716,0.01810391,0.0003434461,0.000026191206,0.0001647423,0.0017823128,0.000111295274,0.002957908],"genre_scores_gemma":[0.9948336,0.00010972554,0.0037360622,0.000048845377,0.000013551976,0.0001413679,0.0008489698,0.000022275772,0.00024564995],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97272533,0.021049252,0.0009476577,0.0015261818,0.0033449964,0.00040661264],"domain_scores_gemma":[0.19266598,0.7608771,0.026016917,0.015253827,0.003438403,0.0017479003],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023386078,0.00052848767,0.00040953193,0.0023233017,0.00046928192,0.0006590846,0.0016225035,0.0011158373,0.0033703058],"category_scores_gemma":[0.32525185,0.00034198977,0.0010649991,0.0024451364,0.0011103711,0.0021224895,0.0007175092,0.0027376448,0.00044024855],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025512625,0.0036648645,0.9343205,0.00032249256,0.0018759419,0.0002444193,0.0008581638,0.009527817,0.0018171798,0.0042275083,0.002209937,0.038379986],"study_design_scores_gemma":[0.00037085998,0.004437483,0.9445165,0.00010601825,0.0008417467,0.00046709366,0.0004971288,0.0401425,0.0018942337,0.0041169315,0.002564096,0.00004535531],"about_ca_topic_score_codex":0.0029491181,"about_ca_topic_score_gemma":0.0037621686,"teacher_disagreement_score":0.023386078,"about_ca_system_score_codex":0.00068028364,"about_ca_system_score_gemma":0.00085872423,"threshold_uncertainty_score":0.12367886},"labels":[],"label_agreement":null},{"id":"W2408469942","doi":"10.1002/smr.1789","title":"A bug reproduction approach based on directed model checking and crash traces","year":2016,"lang":"en","type":"article","venue":"Journal of Software Evolution and Process","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Crash; Computer science; Java; Slicing; Programming language; Model checking; Task (project management); Program slicing; Open source; Debugging; Software; World Wide Web","score_opus":0.01984930774373265,"score_gpt":0.2571976485383381,"score_spread":0.23734834079460546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2408469942","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036674127,0.000099702156,0.955246,0.0000816438,0.00001559137,0.00012612445,0.00006474103,0.0072062705,0.00048580943],"genre_scores_gemma":[0.4594708,0.000088642824,0.53880095,0.000081969716,0.000011323505,0.00011298248,0.00027461222,0.0006186688,0.00053999096],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969266,0.000922625,0.00019892801,0.0005164083,0.0012252637,0.00021009124],"domain_scores_gemma":[0.9907074,0.0042125294,0.0009120928,0.0026884219,0.0012395331,0.00024001069],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027244878,0.0013977275,0.00097419,0.0030019789,0.00047006406,0.0010797781,0.0028503,0.0011825906,0.0012496029],"category_scores_gemma":[0.010486395,0.0008116076,0.0017366954,0.0008074248,0.0012311912,0.0014345188,0.0023662804,0.0014057304,0.00028505828],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006649001,0.00069335476,0.023063553,0.0005376828,0.0004572198,0.0017118336,0.00096732576,0.4954669,0.124404706,0.021267587,0.0019232478,0.32884166],"study_design_scores_gemma":[0.000045442386,0.0001348387,0.0007472506,0.000024966022,0.000088021814,0.00025545835,0.00003845161,0.95920396,0.03189993,0.006527461,0.0009977256,0.00003650862],"about_ca_topic_score_codex":0.0051599317,"about_ca_topic_score_gemma":0.003924857,"teacher_disagreement_score":0.0051599317,"about_ca_system_score_codex":0.00079472316,"about_ca_system_score_gemma":0.0019260158,"threshold_uncertainty_score":0.014408648},"labels":[],"label_agreement":null},{"id":"W2461407631","doi":"10.1002/stvr.1609","title":"Prioritizing manual test cases in rapid release environments","year":2016,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Manitoba","funders":"Mitacs; Lunds Universitet; University of Manitoba","keywords":"Computer science; Prioritization; Test suite; Unit testing; Agile software development; Test (biology); Code coverage; Suite; Test case; Embedded system; Software engineering; Operating system; Software; Engineering; Machine learning","score_opus":0.02512928207923182,"score_gpt":0.25991368093417877,"score_spread":0.23478439885494695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2461407631","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6302709,0.0010893104,0.34764308,0.0005501575,0.00012466336,0.0010467428,0.00025773342,0.008609787,0.010407738],"genre_scores_gemma":[0.8363888,0.0002539735,0.15883611,0.00014934575,0.000057339606,0.00026178895,0.0005662239,0.00064382725,0.0028424999],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9859087,0.005914173,0.0007183689,0.0014201937,0.0051843156,0.0008542521],"domain_scores_gemma":[0.9203357,0.055746738,0.008115263,0.007098893,0.0068033915,0.001899988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068255737,0.0013188218,0.0007074249,0.0033435458,0.00043743447,0.0018425514,0.0021106596,0.0006016444,0.0029470662],"category_scores_gemma":[0.035307616,0.00072936394,0.00059170637,0.0009803995,0.0005939524,0.0016589187,0.0017615611,0.0011154562,0.001017981],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018437615,0.0015501188,0.024738155,0.00077561877,0.0001675085,0.0016581807,0.0018571629,0.0584857,0.07948006,0.0036011073,0.005904751,0.8199379],"study_design_scores_gemma":[0.0012018284,0.009979488,0.08887967,0.0008304685,0.0005010291,0.0063994657,0.0032336428,0.57981807,0.24085876,0.015683284,0.052048925,0.0005654374],"about_ca_topic_score_codex":0.0017800006,"about_ca_topic_score_gemma":0.0022011853,"teacher_disagreement_score":0.0068255737,"about_ca_system_score_codex":0.0006902158,"about_ca_system_score_gemma":0.0011616145,"threshold_uncertainty_score":0.036097527},"labels":[],"label_agreement":null},{"id":"W2462043969","doi":"10.22215/etd/2011-09481","title":"A UML/MARTE model analysis approach for detection of concurrency faults","year":2011,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Library and Archives Canada","funders":"","keywords":"Computer science; Concurrency; Unified Modeling Language; Programming language; Software","score_opus":0.039956633023581635,"score_gpt":0.2926322019118875,"score_spread":0.25267556888830583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2462043969","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072622974,0.00015323613,0.9775599,0.00028249127,0.000056983532,0.00020751818,0.0004682979,0.011214396,0.0027947922],"genre_scores_gemma":[0.08703777,0.00020211627,0.90691537,0.00013244608,0.000019593152,0.00016551797,0.0014780648,0.0008167033,0.0032322905],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99702066,0.00062983716,0.00026590202,0.0003126853,0.0016015931,0.0001693726],"domain_scores_gemma":[0.9963451,0.0012630204,0.00036668917,0.0008204879,0.0011008142,0.0001039246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031312155,0.0011704192,0.0007778262,0.003927738,0.0007651161,0.0025562185,0.0014808176,0.0011111863,0.004492134],"category_scores_gemma":[0.0081076035,0.0009880767,0.0019721582,0.00095795636,0.00057682226,0.0023252456,0.0015252881,0.0015811092,0.0016033906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005540512,0.00083170156,0.010566325,0.00080628274,0.00045600833,0.0012223788,0.0021431902,0.08777136,0.10827241,0.10419387,0.022380035,0.66080236],"study_design_scores_gemma":[0.00007119121,0.000143137,0.0016394885,0.00018032543,0.00020153911,0.00062849367,0.0003460226,0.87669647,0.05473794,0.018979328,0.046294115,0.0000819312],"about_ca_topic_score_codex":0.0066907783,"about_ca_topic_score_gemma":0.013216994,"teacher_disagreement_score":0.0066907783,"about_ca_system_score_codex":0.0009284185,"about_ca_system_score_gemma":0.0026423174,"threshold_uncertainty_score":0.0165596},"labels":[],"label_agreement":null},{"id":"W2479336862","doi":"10.1109/snpd.2016.7515953","title":"A comparative study on black-box testing with open source applications","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Algoma University","funders":"","keywords":"Black box; Open source software; Computer science; Open source; White-box testing; Non-regression testing; Software; Software performance testing; Software testing; Software reliability testing; Reliability engineering; Software engineering; Operating system; Software quality; Software development; Software construction; Engineering; Artificial intelligence","score_opus":0.094093447487158,"score_gpt":0.3420626282076523,"score_spread":0.24796918072049431,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2479336862","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95844394,0.0031376162,0.031961404,0.00012610285,0.000053056992,0.00016082414,0.00009092329,0.0003590954,0.00566709],"genre_scores_gemma":[0.98251134,0.0005943337,0.015596148,0.00003549763,0.000029742825,0.000050143437,0.0002079621,0.0000872909,0.0008876215],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9870367,0.006767755,0.0007011714,0.000872309,0.0040888065,0.00053324504],"domain_scores_gemma":[0.889677,0.089009784,0.0036383914,0.004449133,0.011742043,0.0014836068],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073434943,0.0006809228,0.0005947265,0.0029655627,0.00045750666,0.0009965958,0.0011097816,0.0007074198,0.0017981046],"category_scores_gemma":[0.037323665,0.00020173607,0.00053666625,0.0017382394,0.0007171281,0.0037128455,0.00071646314,0.00046637392,0.00024717514],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008458331,0.0028282069,0.0653747,0.0041487394,0.0004906788,0.0013043828,0.005977062,0.023417529,0.14364609,0.0070398627,0.0019257078,0.7353887],"study_design_scores_gemma":[0.00077030895,0.059348755,0.3886321,0.001846945,0.0016753604,0.005207131,0.012036946,0.24941342,0.24238133,0.00912962,0.029107299,0.0004508393],"about_ca_topic_score_codex":0.001099144,"about_ca_topic_score_gemma":0.000980166,"teacher_disagreement_score":0.0073434943,"about_ca_system_score_codex":0.0006114901,"about_ca_system_score_gemma":0.00041765842,"threshold_uncertainty_score":0.0388366},"labels":[],"label_agreement":null},{"id":"W2480723093","doi":"10.1007/b136676","title":"Testing of Communicating Systems","year":2005,"lang":"en","type":"book","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Tomsk State University; Tsinghua University; Georg-August-Universität Göttingen; Ohio State University; Universidad Complutense de Madrid; University of Ottawa","keywords":"Computer science","score_opus":0.03524634707782208,"score_gpt":0.27733617445007575,"score_spread":0.24208982737225365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2480723093","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035556413,0.007167939,0.7316118,0.0010620662,0.0005499899,0.00013603336,0.00021577872,0.0041240063,0.21957596],"genre_scores_gemma":[0.62516075,0.00742505,0.2231585,0.00044764343,0.0006283887,0.00023967336,0.0010463616,0.0011709635,0.14072266],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989793,0.00021190096,0.000039146482,0.00010641421,0.0006072327,0.000056018358],"domain_scores_gemma":[0.99745303,0.0016729205,0.00008477407,0.00046765173,0.00025052155,0.00007104639],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00055078376,0.0009933212,0.00065672165,0.0010964382,0.0004502173,0.0013260255,0.001219353,0.00072595844,0.0115095815],"category_scores_gemma":[0.0039047883,0.0004888067,0.00045862736,0.0011271446,0.0013525801,0.0018719069,0.0009377827,0.0013215742,0.0022943146],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015018777,0.0001395263,0.00083835184,0.0005063803,0.000048337464,0.00044891695,0.00059340004,0.012977777,0.020571267,0.3034498,0.020994095,0.639282],"study_design_scores_gemma":[0.000075206524,0.00036820272,0.0016692929,0.00042664487,0.000082914295,0.0016490319,0.00016130757,0.09249294,0.053375896,0.7164483,0.13320524,0.000045054105],"about_ca_topic_score_codex":0.0003623033,"about_ca_topic_score_gemma":0.00034401764,"teacher_disagreement_score":0.0115095815,"about_ca_system_score_codex":0.00036181175,"about_ca_system_score_gemma":0.00037272886,"threshold_uncertainty_score":0.03850335},"labels":[],"label_agreement":null},{"id":"W2485697085","doi":"10.1109/icst.2016.32","title":"Atrina: Inferring Unit Oracles from GUI Test Cases","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Intel Corporation","keywords":"JavaScript; Computer science; Unit testing; Leverage (statistics); Programming language; Assertion; Test suite; Test case; Software testing; Code (set theory); Artificial intelligence; Software; Machine learning","score_opus":0.048376541193908905,"score_gpt":0.2788364857413657,"score_spread":0.2304599445474568,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2485697085","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13715175,0.00060308096,0.7644692,0.00025137144,0.000053925687,0.0003627258,0.0034073358,0.091415346,0.0022852886],"genre_scores_gemma":[0.5531894,0.00024408741,0.43271875,0.00019318906,0.000039694765,0.00027742572,0.010159856,0.0018103167,0.0013672392],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9959669,0.0011747152,0.00038117604,0.0010998152,0.0011491742,0.00022826144],"domain_scores_gemma":[0.9755574,0.015470987,0.0029788106,0.003180252,0.0023181739,0.0004944928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033196735,0.0024212396,0.0007583177,0.005847698,0.00033793465,0.0017032275,0.0027013987,0.0015616422,0.0028295442],"category_scores_gemma":[0.029398713,0.0005620407,0.0011818936,0.0011898209,0.00065774017,0.001991177,0.0013754148,0.0009096789,0.0018262393],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086392666,0.0008979403,0.13363348,0.0014649424,0.00037429828,0.002543394,0.0008216148,0.10269189,0.04213403,0.006790571,0.015433278,0.6923507],"study_design_scores_gemma":[0.00006103694,0.00021287065,0.008799715,0.000140024,0.000089191606,0.00076117064,0.00015906736,0.94032973,0.040074494,0.005069066,0.0042597204,0.00004390146],"about_ca_topic_score_codex":0.0052734464,"about_ca_topic_score_gemma":0.007899358,"teacher_disagreement_score":0.005847698,"about_ca_system_score_codex":0.0006904698,"about_ca_system_score_gemma":0.0012967242,"threshold_uncertainty_score":0.01755637},"labels":[],"label_agreement":null},{"id":"W2512737610","doi":"","title":"Global Applications of Array-Based Techniques II Posters","year":2015,"lang":"en","type":"article","venue":"2015 AGU Fall Meeting","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science","score_opus":0.026468031843469682,"score_gpt":0.292040474495669,"score_spread":0.2655724426521993,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2512737610","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049665123,0.013691342,0.5466462,0.012624745,0.01619659,0.00033486035,0.0011483068,0.008887585,0.3508053],"genre_scores_gemma":[0.33047983,0.007825773,0.17508394,0.0012485434,0.0070003252,0.00023764599,0.0015669117,0.0032998144,0.47325715],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99942905,0.00013096376,0.000018354187,0.00013502194,0.00020345926,0.00008307885],"domain_scores_gemma":[0.9988431,0.00024340182,0.00004422905,0.00023475463,0.00041467548,0.0002197971],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001068581,0.00114607,0.00050193106,0.0012065277,0.00086085964,0.0023072043,0.00069715653,0.0009738801,0.06529575],"category_scores_gemma":[0.0014549785,0.00035868515,0.0006990743,0.0011950765,0.000562704,0.0017181049,0.0018159109,0.0015320997,0.011204552],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064563606,0.00018291223,0.0013698658,0.0003619242,0.000072002724,0.0003429108,0.00035298473,0.007817131,0.047879394,0.045720454,0.1994045,0.6958502],"study_design_scores_gemma":[0.00018094249,0.0012700597,0.007705465,0.00043680353,0.0001832125,0.0014559646,0.00047065783,0.04203939,0.08543527,0.07579784,0.78489053,0.00013389075],"about_ca_topic_score_codex":0.00044576245,"about_ca_topic_score_gemma":0.0010104106,"teacher_disagreement_score":0.06529575,"about_ca_system_score_codex":0.0005058033,"about_ca_system_score_gemma":0.00040113117,"threshold_uncertainty_score":0.21843606},"labels":[],"label_agreement":null},{"id":"W2516023085","doi":"10.4204/eptcs.224.6","title":"Fault Localization in Web Applications via Model Finding","year":2016,"lang":"en","type":"article","venue":"Electronic Proceedings in Theoretical Computer Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Computer science; Satisfiability; Programming language; Focus (optics); Set (abstract data type); Object (grammar); Fault (geology); Theoretical computer science; Specification language; Fault model; Algorithm; Artificial intelligence; Engineering","score_opus":0.007970434223973321,"score_gpt":0.2536372894738852,"score_spread":0.2456668552499119,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2516023085","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01000361,0.00007021081,0.98729414,0.0001683706,0.000009044444,0.000039388113,0.000022391576,0.0017965329,0.0005963717],"genre_scores_gemma":[0.43491518,0.00026363897,0.5621947,0.00014174916,0.000020487909,0.00020259117,0.0001556197,0.00037423405,0.0017317383],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975712,0.00078935997,0.00014927347,0.0004219818,0.0008538598,0.00021443455],"domain_scores_gemma":[0.9952604,0.0029495393,0.00040527905,0.0009865552,0.00033682355,0.00006139736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018750294,0.00086220435,0.00082774035,0.001340796,0.00066757336,0.0020306914,0.002362154,0.0016344977,0.0017872084],"category_scores_gemma":[0.009442505,0.0006525957,0.0015100145,0.00095038675,0.0020430505,0.0033598375,0.0029497198,0.0019846503,0.0004336474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021776043,0.00030044233,0.0038448586,0.000610587,0.0001502759,0.0013009663,0.0010969781,0.52658725,0.039135214,0.19994469,0.002433156,0.2243778],"study_design_scores_gemma":[0.000025956268,0.000054923432,0.00014820279,0.000034407854,0.000043879783,0.00027842564,0.00008816108,0.8784713,0.027263442,0.0908831,0.0026872514,0.000020906053],"about_ca_topic_score_codex":0.0022772993,"about_ca_topic_score_gemma":0.00258514,"teacher_disagreement_score":0.002362154,"about_ca_system_score_codex":0.0011271149,"about_ca_system_score_gemma":0.0014809502,"threshold_uncertainty_score":0.009916186},"labels":[],"label_agreement":null},{"id":"W2517343942","doi":"10.1145/2975961.2975962","title":"Mining timed regular expressions from system traces","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Debugging; Computer science; Rotation formalisms in three dimensions; Event (particle physics); Anomaly (physics); Anomaly detection; Programming language; Temporal logic; Real-time computing; Data mining","score_opus":0.019495748284716483,"score_gpt":0.23320831570258288,"score_spread":0.2137125674178664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2517343942","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30254722,0.001047037,0.67827725,0.0006628742,0.000086625805,0.0006487971,0.008173473,0.006071033,0.00248566],"genre_scores_gemma":[0.64859766,0.00089315925,0.32817098,0.00012351408,0.00007524946,0.0005644504,0.019271499,0.0004605895,0.0018429941],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978257,0.00037685543,0.00029060998,0.0005769224,0.0007328173,0.00019702999],"domain_scores_gemma":[0.9906882,0.005788312,0.0013205017,0.0008956571,0.0011111,0.00019627748],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012378867,0.0012123112,0.0008140144,0.003981519,0.00046092673,0.0012680038,0.0013522375,0.0008074301,0.00089428964],"category_scores_gemma":[0.012351244,0.00056385587,0.0015321749,0.0023210193,0.0006258398,0.0015753575,0.0007149009,0.0009903164,0.0005166745],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009902582,0.000857223,0.06882895,0.0023943176,0.0005940099,0.005959408,0.0014989767,0.3345961,0.050395515,0.028599838,0.009848752,0.49543667],"study_design_scores_gemma":[0.00009703118,0.00020317525,0.0070524705,0.00013292764,0.00013925518,0.00086706586,0.00046047746,0.93082285,0.018440599,0.03507605,0.0066589992,0.000049035425],"about_ca_topic_score_codex":0.0049469485,"about_ca_topic_score_gemma":0.007032933,"teacher_disagreement_score":0.0049469485,"about_ca_system_score_codex":0.00067203416,"about_ca_system_score_gemma":0.0020812398,"threshold_uncertainty_score":0.009836316},"labels":[],"label_agreement":null},{"id":"W2519113869","doi":"10.1109/ivsw.2016.7566604","title":"Revision debug with non-linear version history in regression verification","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Debugging; Computer science; Bottleneck; Preprocessor; Ranking (information retrieval); Regression testing; Software bug; Correctness; Graph; Software engineering; Programming language; Theoretical computer science; Embedded system; Information retrieval; Software system","score_opus":0.02317134808403726,"score_gpt":0.25303797536745515,"score_spread":0.2298666272834179,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2519113869","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11365461,0.0008455589,0.87385213,0.00027972143,0.000071064664,0.0001589459,0.00013820526,0.008185014,0.0028146643],"genre_scores_gemma":[0.71510774,0.00021472048,0.28120333,0.000099829354,0.000065672924,0.00008547346,0.00019953927,0.00039625962,0.0026274242],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949945,0.001805487,0.00030679826,0.0007128401,0.0017637975,0.0004164579],"domain_scores_gemma":[0.9827611,0.009806947,0.0018445001,0.0035592257,0.0015877513,0.000440442],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005378988,0.0008967713,0.0009623211,0.0039277677,0.0007984297,0.0017063502,0.0014024302,0.00078135566,0.002448521],"category_scores_gemma":[0.025147505,0.00077057176,0.0007080204,0.0022060438,0.0014460641,0.0039224434,0.0019604175,0.0016931696,0.0005704619],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012219342,0.00029867506,0.01845285,0.00020497521,0.000081629514,0.0003470019,0.0006430696,0.17970626,0.015546974,0.052786347,0.0030782092,0.72763205],"study_design_scores_gemma":[0.00008240986,0.00052662846,0.0021188254,0.000057816374,0.00006278058,0.0002656552,0.00006331896,0.9214122,0.020054214,0.052061237,0.0032075243,0.0000873171],"about_ca_topic_score_codex":0.002462406,"about_ca_topic_score_gemma":0.0036743644,"teacher_disagreement_score":0.005378988,"about_ca_system_score_codex":0.00092091673,"about_ca_system_score_gemma":0.0017263383,"threshold_uncertainty_score":0.028447151},"labels":[],"label_agreement":null},{"id":"W2521099350","doi":"10.1007/s11219-016-9340-8","title":"An empirical study on the effects of code visibility on program testability","year":2016,"lang":"en","type":"article","venue":"Software Quality Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Fundamental Research Funds for the Central Universities","keywords":"Testability; Visibility; Computer science; Code coverage; Artifact (error); Code (set theory); Software; Software quality; Software engineering; Software metric; Programming language; Software development; Reliability engineering; Engineering; Artificial intelligence; Set (abstract data type)","score_opus":0.06644119961895152,"score_gpt":0.4136435348790582,"score_spread":0.3472023352601067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2521099350","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99830437,0.00013495814,0.00050927105,0.000049445378,0.0000031054062,0.000015194567,0.00004551435,0.000010242208,0.0009278771],"genre_scores_gemma":[0.999316,0.00004257713,0.00041934723,0.000010820271,0.0000042263737,0.000008163457,0.00004550693,0.0000073275082,0.00014620244],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9906233,0.0053901975,0.00068872934,0.0007281698,0.0020978444,0.00047171573],"domain_scores_gemma":[0.2573614,0.68324775,0.037909564,0.009921348,0.007607208,0.003952626],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.00835709,0.00033466346,0.00030681907,0.0014439813,0.00047377165,0.0011034086,0.0007606927,0.00071892777,0.0032087197],"category_scores_gemma":[0.18259357,0.00034202874,0.00063265796,0.0013620445,0.0013440972,0.002047631,0.0011265568,0.0015645719,0.00020370493],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025687907,0.0044306405,0.9499137,0.00023190475,0.00025166085,0.00028038246,0.0027617882,0.0009941929,0.0034059407,0.0008441043,0.00027558784,0.034041267],"study_design_scores_gemma":[0.00008778366,0.0033089437,0.98966557,0.00004336519,0.00019960129,0.00031521736,0.0010882805,0.0025534031,0.0018670624,0.0003961188,0.00045524907,0.000019411727],"about_ca_topic_score_codex":0.0026062091,"about_ca_topic_score_gemma":0.0033874533,"teacher_disagreement_score":0.9916429,"about_ca_system_score_codex":0.0005622097,"about_ca_system_score_gemma":0.0010378022,"threshold_uncertainty_score":0.044197083},"labels":[],"label_agreement":null},{"id":"W2528843943","doi":"","title":"Towards a formalization of viewpoints testing","year":2002,"lang":"en","type":"article","venue":"Kent Academic Repository (University of Kent)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; Natural Sciences and Engineering Research Council of Canada","keywords":"Viewpoints; Computer science","score_opus":0.03945148300212559,"score_gpt":0.21778578672480411,"score_spread":0.17833430372267853,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2528843943","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015671019,0.000112069385,0.9958348,0.00040587469,0.000042692067,0.00005566439,0.000036297377,0.0005944963,0.0013509353],"genre_scores_gemma":[0.11449159,0.00037880478,0.88189536,0.00041048188,0.00015427255,0.00020099914,0.00028200253,0.00054284575,0.001643694],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97966033,0.00852779,0.002472565,0.0023323444,0.00567789,0.001329116],"domain_scores_gemma":[0.9495938,0.028808886,0.0025356025,0.011021926,0.0070082378,0.0010315987],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01568614,0.0014002429,0.0014414982,0.0030938159,0.0019155364,0.007879975,0.006354875,0.0031424297,0.004200857],"category_scores_gemma":[0.044495635,0.0029165235,0.0037829287,0.0023256727,0.010630337,0.013137323,0.007648389,0.010280439,0.0010656395],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000042197305,0.000092450326,0.00088719645,0.00020189909,0.000041498788,0.00015557707,0.0008590184,0.022431497,0.0016430912,0.93078285,0.0016717579,0.041190937],"study_design_scores_gemma":[0.00006020279,0.00005422332,0.0002479445,0.00030571077,0.00010677516,0.00018232763,0.0002777174,0.15680933,0.0039683515,0.81828654,0.01963763,0.00006322317],"about_ca_topic_score_codex":0.0077132075,"about_ca_topic_score_gemma":0.008662876,"teacher_disagreement_score":0.01568614,"about_ca_system_score_codex":0.003365344,"about_ca_system_score_gemma":0.0060957563,"threshold_uncertainty_score":0.08295721},"labels":[],"label_agreement":null},{"id":"W2535681473","doi":"10.1109/mutation.2006.10","title":"Mutation Operators for Concurrent Java (J2SE 5.0)","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":94,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Concurrency; Java; Programming language; Java concurrency; Operating system; Semaphore; Real time Java","score_opus":0.0165711930763524,"score_gpt":0.27212979571160767,"score_spread":0.25555860263525526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2535681473","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034449104,0.00036722148,0.935554,0.00042175574,0.0003340103,0.00046398197,0.0003285188,0.022793053,0.005288229],"genre_scores_gemma":[0.23383972,0.00041230366,0.75418437,0.0007362658,0.00013031681,0.00058033975,0.00059671176,0.0051386994,0.004381282],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9964269,0.00055331044,0.00047365215,0.00041604662,0.001873109,0.00025701575],"domain_scores_gemma":[0.9928798,0.003516955,0.0010718096,0.0009537823,0.0013430887,0.00023467069],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004192478,0.0010011342,0.000583102,0.001575425,0.0010025105,0.0016188117,0.0015937247,0.0010769853,0.0025560693],"category_scores_gemma":[0.011410913,0.000668151,0.001609945,0.0006652701,0.0015440307,0.0020761606,0.0023277048,0.0020482498,0.0006686154],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010138208,0.0007572261,0.016416753,0.0017952373,0.00026929087,0.0028239065,0.0026698268,0.018761408,0.14931211,0.2631988,0.035753887,0.5072277],"study_design_scores_gemma":[0.0006978325,0.0013522077,0.010307836,0.0006692333,0.00060998613,0.009678012,0.0005739609,0.23680449,0.2253244,0.21506096,0.29827592,0.00064520095],"about_ca_topic_score_codex":0.001794365,"about_ca_topic_score_gemma":0.0022751668,"teacher_disagreement_score":0.004192478,"about_ca_system_score_codex":0.00083473313,"about_ca_system_score_gemma":0.0016249176,"threshold_uncertainty_score":0.022172153},"labels":[],"label_agreement":null},{"id":"W2544310856","doi":"10.1109/icm.2011.6177404","title":"Performance analysis of constraint solvers for coverage directed test generation","year":2011,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Constraint (computer-aided design); Test (biology); Mathematics","score_opus":0.05718253183475186,"score_gpt":0.25378960447336724,"score_spread":0.19660707263861538,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2544310856","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.68695295,0.0023040406,0.28010428,0.00071998494,0.00009662472,0.00036620832,0.0014259964,0.009224169,0.018805683],"genre_scores_gemma":[0.830593,0.0004028563,0.16437751,0.00015360065,0.000025409145,0.00019445627,0.0021928437,0.0005442731,0.0015160331],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9908187,0.004996568,0.0004739357,0.00067628355,0.0023986227,0.00063572533],"domain_scores_gemma":[0.930917,0.06111887,0.001760803,0.0020623524,0.003560463,0.0005804004],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004863513,0.0010328493,0.0007609193,0.0014237983,0.00037296003,0.0012033272,0.0011229449,0.0010352806,0.0056865704],"category_scores_gemma":[0.035872843,0.00034685608,0.0005504969,0.0020648255,0.00055292697,0.0012209429,0.0006132372,0.0007783444,0.0006813256],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023692788,0.00045459255,0.0055431062,0.0005968609,0.00027516662,0.00020129432,0.0001380006,0.7711307,0.01345114,0.00814912,0.0044340515,0.19325662],"study_design_scores_gemma":[0.00013367171,0.00030022408,0.00060749357,0.000016004933,0.000037127866,0.00007390683,0.000032857,0.9901625,0.007027181,0.00096307317,0.0006346023,0.00001134962],"about_ca_topic_score_codex":0.0062709837,"about_ca_topic_score_gemma":0.0050800783,"teacher_disagreement_score":0.0062709837,"about_ca_system_score_codex":0.0013411415,"about_ca_system_score_gemma":0.0024857987,"threshold_uncertainty_score":0.025721014},"labels":[],"label_agreement":null},{"id":"W2546974490","doi":"10.1016/bs.adcom.2017.09.003","title":"Optimizing the Symbolic Execution of Evolving Rhapsody Statecharts","year":2017,"lang":"en","type":"book-chapter","venue":"Advances in computers","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Programming language; Symbolic execution; The Symbolic; Psychology; Software; Psychoanalysis","score_opus":0.019843225956895267,"score_gpt":0.2774101269308896,"score_spread":0.2575669009739944,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2546974490","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07232985,0.00026715826,0.9062642,0.00013825191,0.00007970158,0.00008481584,0.00015160516,0.009230921,0.011453559],"genre_scores_gemma":[0.55475193,0.00030431146,0.43629384,0.00006430571,0.000025327421,0.000118772856,0.00050046627,0.001952491,0.0059885806],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99908304,0.00027080067,0.00004715995,0.000127064,0.0003494919,0.00012250339],"domain_scores_gemma":[0.9979498,0.0013770871,0.00010313172,0.00029257,0.00023230012,0.000045112738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008307976,0.0009651143,0.0006916114,0.0005974677,0.00046689188,0.0010526662,0.0018341203,0.00068780885,0.008075544],"category_scores_gemma":[0.0044787857,0.0004918951,0.0007759209,0.0007637802,0.0011336068,0.0014503885,0.0013388105,0.0013205808,0.0010335058],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035481222,0.00012306811,0.0012974143,0.00054289633,0.00007245682,0.00028686383,0.00046833293,0.5314931,0.04034675,0.07886871,0.0037196095,0.34242594],"study_design_scores_gemma":[0.0000473741,0.00008909419,0.00023011409,0.00005064683,0.000037129998,0.000068064226,0.00004911285,0.93345,0.032403857,0.029055344,0.004494558,0.000024694362],"about_ca_topic_score_codex":0.0032250446,"about_ca_topic_score_gemma":0.00441062,"teacher_disagreement_score":0.008075544,"about_ca_system_score_codex":0.0009779717,"about_ca_system_score_gemma":0.00090515404,"threshold_uncertainty_score":0.027015328},"labels":[],"label_agreement":null},{"id":"W2548701964","doi":"10.1145/2994291.2994294","title":"PredSym: estimating software testing budget for a bug-free release","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Overfitting; Symbolic execution; Regression testing; Software; Code (set theory); Software performance testing; Code coverage; Software bug; Concolic testing; Software testing; Programming language; Software engineering; Machine learning; Software system; Software construction; Artificial neural network","score_opus":0.0332152410180936,"score_gpt":0.2673196379326332,"score_spread":0.2341043969145396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2548701964","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4828615,0.0008648992,0.40805045,0.0004979846,0.00010770431,0.00032212335,0.011181044,0.09010438,0.0060098805],"genre_scores_gemma":[0.68850523,0.00022932579,0.29564995,0.00007005076,0.00002622668,0.0003774406,0.0107508525,0.0022461726,0.0021447197],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986986,0.00030232288,0.000107631975,0.0003024913,0.00047397244,0.00011495419],"domain_scores_gemma":[0.9852962,0.009047364,0.0022170355,0.0014482989,0.0016656502,0.0003254801],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023684923,0.0016276621,0.0006436822,0.002449127,0.00027502835,0.0006933728,0.0011320151,0.00062696036,0.00400711],"category_scores_gemma":[0.022995984,0.0005942182,0.00052879506,0.0010175883,0.00035749766,0.0015049027,0.0006827784,0.00078353175,0.0015930078],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014502206,0.0004145775,0.105974555,0.0010012259,0.00018858786,0.00039545263,0.00038007792,0.4098914,0.021044161,0.0043056067,0.028814072,0.42614016],"study_design_scores_gemma":[0.000048211125,0.00023670089,0.011603706,0.00004753244,0.000024743797,0.00011167702,0.00005202282,0.97316474,0.010774025,0.0011102103,0.0027951358,0.000031250107],"about_ca_topic_score_codex":0.003858386,"about_ca_topic_score_gemma":0.0061373417,"teacher_disagreement_score":0.00400711,"about_ca_system_score_codex":0.0007232433,"about_ca_system_score_gemma":0.00120463,"threshold_uncertainty_score":0.013405085},"labels":[],"label_agreement":null},{"id":"W2549749398","doi":"10.14288/1.0319084","title":"A study of the influence of assertions and mutants on test suite effectiveness","year":2016,"lang":"en","type":"article","venue":"cIRcle (University of British Columbia)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Suite; Test suite; Test (biology); Computer science; Biology; Test case; History; Archaeology","score_opus":0.009772553224793704,"score_gpt":0.1978724058906307,"score_spread":0.188099852665837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2549749398","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9796978,0.0021515489,0.015750589,0.00018185421,0.000025432695,0.000057403067,0.00023649845,0.00038977133,0.0015089464],"genre_scores_gemma":[0.99199474,0.00023145982,0.006957995,0.000045708028,0.000033703054,0.00004493532,0.0003873699,0.00013587331,0.00016817363],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9603591,0.017703716,0.003255044,0.0047429404,0.012187903,0.0017513279],"domain_scores_gemma":[0.20420896,0.7391538,0.030056834,0.013790929,0.0106387995,0.0021506997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027267855,0.0016243932,0.0012642057,0.00511095,0.0004441026,0.002134318,0.0013055968,0.0012841461,0.00094480853],"category_scores_gemma":[0.2889583,0.0006432576,0.0015688392,0.0026858742,0.0016613932,0.004247333,0.0011531542,0.0018277812,0.00026884326],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029639187,0.0011497793,0.57304287,0.0010194512,0.0023958248,0.0014206235,0.0010812379,0.17792484,0.03972494,0.0032992552,0.0010765954,0.1949007],"study_design_scores_gemma":[0.00018023758,0.0071313935,0.4929965,0.0002534858,0.001761377,0.0020840152,0.00070881855,0.45868084,0.031157605,0.0030011176,0.0018542847,0.00019019679],"about_ca_topic_score_codex":0.0023773953,"about_ca_topic_score_gemma":0.0016194846,"teacher_disagreement_score":0.027267855,"about_ca_system_score_codex":0.0011881544,"about_ca_system_score_gemma":0.0010382793,"threshold_uncertainty_score":0.14420795},"labels":[],"label_agreement":null},{"id":"W2549903643","doi":"10.1109/hldvt.2016.7748249","title":"Accelerating assertion assessment using GPUs","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; McGill University; CMC Microsystems (Canada)","funders":"","keywords":"Assertion; Correctness; Computer science; Parallelism (grammar); Key (lock); Programming language; Formal verification; Parallel computing; Theoretical computer science; Operating system","score_opus":0.07412440821616174,"score_gpt":0.3391528525172591,"score_spread":0.26502844430109734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2549903643","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.106762424,0.00030573335,0.8747839,0.00039915592,0.00015477184,0.00008858562,0.00012851978,0.010508398,0.0068685124],"genre_scores_gemma":[0.6102594,0.00011925529,0.38613698,0.0001778667,0.000023330787,0.00009026637,0.00019223393,0.00047412553,0.002526505],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984621,0.00050868327,0.0000829252,0.00024345706,0.00056454004,0.00013831844],"domain_scores_gemma":[0.99422085,0.0030731596,0.00035609078,0.0014661065,0.0007541505,0.00012966736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011911937,0.00066339754,0.00052551244,0.000633535,0.00035212343,0.00094660616,0.0019784444,0.00073083513,0.0060603563],"category_scores_gemma":[0.007735,0.0004047486,0.0006040262,0.00051292655,0.0008865847,0.0019364938,0.0012374423,0.001347404,0.0010450099],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014710758,0.0003266756,0.0153469,0.0005367403,0.00021147229,0.0007564434,0.00070319214,0.2643443,0.12411085,0.08024891,0.009768987,0.50217444],"study_design_scores_gemma":[0.000050628205,0.0001297524,0.00069143786,0.000024797448,0.000032805445,0.00013526191,0.000051304196,0.94629085,0.028036142,0.01883596,0.005704496,0.000016585149],"about_ca_topic_score_codex":0.004040157,"about_ca_topic_score_gemma":0.0055463286,"teacher_disagreement_score":0.0060603563,"about_ca_system_score_codex":0.00081982504,"about_ca_system_score_gemma":0.0011274022,"threshold_uncertainty_score":0.020273924},"labels":[],"label_agreement":null},{"id":"W2550120381","doi":"10.48550/arxiv.1611.01855","title":"Neuro-Symbolic Program Synthesis","year":2016,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":105,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Correctness; Computer science; Construct (python library); Program synthesis; Set (abstract data type); Representation (politics); Artificial neural network; Domain (mathematical analysis); Task (project management); String (physics); Theoretical computer science; Regular expression; Algorithm; Artificial intelligence; Programming language; Mathematics","score_opus":0.06483697021880241,"score_gpt":0.18710201138523988,"score_spread":0.12226504116643747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2550120381","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01629951,0.00058646576,0.9708693,0.00027907576,0.000047496862,0.0000701491,0.00014353893,0.001803094,0.009901391],"genre_scores_gemma":[0.41043466,0.000990224,0.5809899,0.00016114737,0.000040006846,0.00037521875,0.00046062912,0.00026951946,0.0062787216],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9997224,0.000054122116,0.000019376395,0.00007253205,0.00010715577,0.000024291072],"domain_scores_gemma":[0.9995647,0.0002357416,0.000033731394,0.00008220335,0.00007155047,0.000012144494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00040388448,0.0005309367,0.00041026558,0.00047440748,0.00023660072,0.00058532314,0.00095075305,0.0005739598,0.003561409],"category_scores_gemma":[0.0013786474,0.00025872645,0.00063462363,0.00041575922,0.0008823032,0.00067899656,0.0006456896,0.00080456916,0.0005816882],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009020478,0.000053756485,0.00057379744,0.0004174943,0.000058552458,0.00013844208,0.00012046159,0.6299821,0.019170823,0.10432341,0.0020575302,0.24301346],"study_design_scores_gemma":[0.000014447185,0.000031143987,0.000082456994,0.00002612933,0.000011926914,0.000043694337,0.000013745074,0.96113956,0.0069606085,0.025620744,0.006049076,0.0000065819336],"about_ca_topic_score_codex":0.0019309868,"about_ca_topic_score_gemma":0.0028867044,"teacher_disagreement_score":0.003561409,"about_ca_system_score_codex":0.0007548249,"about_ca_system_score_gemma":0.0010544148,"threshold_uncertainty_score":0.011914074},"labels":[],"label_agreement":null},{"id":"W2562367443","doi":"10.22360/summersim.2016.scsc.019","title":"On Simulation-based Metrics that Characterize the Behavior of RTL Errors","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Debugging; Debugger; Process (computing); Pruning; Software bug; Programming language; Reliability engineering; Software","score_opus":0.06359309796219473,"score_gpt":0.29865548424512556,"score_spread":0.23506238628293083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2562367443","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17326704,0.00063994806,0.8195915,0.0001957874,0.00003453866,0.00024511243,0.0005663341,0.002180076,0.003279649],"genre_scores_gemma":[0.77974427,0.00043939377,0.21721236,0.00008198062,0.000030594158,0.00033443084,0.0011148631,0.00038970818,0.0006524871],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.991963,0.0028033177,0.0006297118,0.0007926411,0.0034940627,0.00031721414],"domain_scores_gemma":[0.9451517,0.03265366,0.009773819,0.0067848633,0.0051681334,0.00046775836],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052685062,0.0018438507,0.0010269197,0.0059956564,0.0004335381,0.0016061686,0.0009041736,0.00110978,0.0009908528],"category_scores_gemma":[0.051635906,0.00044557275,0.0007963793,0.0041097812,0.0014773231,0.0026760076,0.00097606576,0.0012705281,0.0003732156],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032230926,0.0004183238,0.042666256,0.00047133045,0.00029743506,0.00022321155,0.00041992174,0.77318984,0.031005362,0.029193137,0.0010192744,0.120773606],"study_design_scores_gemma":[0.000018206838,0.00051638903,0.012429866,0.00009966867,0.000049935028,0.00040839514,0.000083888866,0.9600515,0.015035512,0.009310196,0.0019339388,0.000062543855],"about_ca_topic_score_codex":0.002173913,"about_ca_topic_score_gemma":0.0024370155,"teacher_disagreement_score":0.0059956564,"about_ca_system_score_codex":0.0013362837,"about_ca_system_score_gemma":0.0013381769,"threshold_uncertainty_score":0.027862847},"labels":[],"label_agreement":null},{"id":"W2584970301","doi":"10.1109/icosst.2016.7838329","title":"Safe regression test suite optimization: A review","year":2016,"lang":"en","type":"review","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Test suite; Regression testing; Computer science; Suite; Reduction (mathematics); Test case; Fuzzy logic; Machine learning; Artificial intelligence; Regression analysis; Mathematics; Software","score_opus":0.057840892821870986,"score_gpt":0.36391140119092785,"score_spread":0.30607050836905686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2584970301","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00049997284,0.99424005,0.003393311,0.0002762931,0.00010990254,0.00003318999,0.000053746535,0.00005394372,0.0013396591],"genre_scores_gemma":[0.0055301846,0.9864118,0.0066401083,0.00022420744,0.00017325903,0.000059823524,0.00023908584,0.00003488397,0.00068657444],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983423,0.00029230572,0.00028045964,0.00024561002,0.0007765644,0.000062714025],"domain_scores_gemma":[0.9931484,0.004512825,0.00061938184,0.00020554403,0.0013840491,0.00012982577],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002478387,0.0015474446,0.00191996,0.0047342456,0.00027589235,0.001054496,0.0030235238,0.0014348625,0.002800502],"category_scores_gemma":[0.007053512,0.00073148665,0.0012101457,0.005501502,0.000671858,0.0016628008,0.0007142363,0.0010173358,0.0013622865],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044623783,0.00009319529,0.00050591346,0.017587077,0.00013009323,0.00011859966,0.000045436478,0.0025713842,0.0008998942,0.0025554306,0.010538881,0.96490955],"study_design_scores_gemma":[0.00007371277,0.0006811276,0.003746599,0.022796785,0.00083204685,0.0032768135,0.00017440766,0.0051989215,0.004882771,0.007979947,0.9502207,0.00013611099],"about_ca_topic_score_codex":0.0028817467,"about_ca_topic_score_gemma":0.0029721938,"teacher_disagreement_score":0.0047342456,"about_ca_system_score_codex":0.00091459276,"about_ca_system_score_gemma":0.0023185892,"threshold_uncertainty_score":0.013107121},"labels":[],"label_agreement":null},{"id":"W2587858343","doi":"10.1109/aspdac.2017.7858329","title":"An extensible perceptron framework for revision RTL debug automation","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Debugging; Computer science; Leverage (statistics); Perceptron; Algorithmic program debugging; Cluster analysis; Automation; Machine learning; Data mining; Artificial intelligence; Programming language; Artificial neural network; Engineering","score_opus":0.04757200302978046,"score_gpt":0.37120749215319254,"score_spread":0.3236354891234121,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2587858343","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0039564506,0.00022065573,0.9904428,0.00015197006,0.000048450092,0.000040851006,0.00012644297,0.0042531,0.00075918256],"genre_scores_gemma":[0.3276458,0.00045171182,0.66350734,0.0003979857,0.00015216798,0.00034595342,0.0008395636,0.0005351325,0.006124304],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998801,0.00041485895,0.00008843991,0.0002681158,0.0003077042,0.00011984535],"domain_scores_gemma":[0.9976775,0.001349598,0.00017212871,0.00027182134,0.00042552486,0.000103428676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029594488,0.0012639505,0.0009120461,0.00111564,0.00045062022,0.0012589475,0.0029057858,0.0015064549,0.004702805],"category_scores_gemma":[0.0066747046,0.0006909124,0.0012857505,0.0007869352,0.0006685344,0.0016451653,0.0013478299,0.002482182,0.0015703308],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024878286,0.00019489926,0.0019012279,0.00017018752,0.00016389671,0.00022646041,0.00013082934,0.7052406,0.0034555795,0.015374334,0.0050797416,0.26781335],"study_design_scores_gemma":[0.000007493801,0.000019131237,0.00009669676,0.000008205987,0.00000944244,0.000018022492,0.00000432266,0.990625,0.000516233,0.007850659,0.00083898025,0.000005723499],"about_ca_topic_score_codex":0.0071058627,"about_ca_topic_score_gemma":0.012005784,"teacher_disagreement_score":0.0071058627,"about_ca_system_score_codex":0.00104499,"about_ca_system_score_gemma":0.0015144207,"threshold_uncertainty_score":0.015732467},"labels":[],"label_agreement":null},{"id":"W2592345979","doi":"10.1109/access.2017.2676161","title":"Patch-Related Vulnerability Detection Based on Symbolic Execution","year":2017,"lang":"en","type":"article","venue":"IEEE Access","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"St. Francis Xavier University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Buffer overflow; Software bug; Symbolic execution; Fuzz testing; Memory safety; Software; Software security assurance; Memory leak; Security bug; Vulnerability (computing); Append; Malware; Static analysis; Operating system; Computer security; Programming language; Memory management; Cloud computing","score_opus":0.04551933161146442,"score_gpt":0.3414228419462515,"score_spread":0.29590351033478707,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2592345979","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20166668,0.00037383402,0.77165043,0.0001736773,0.000031413507,0.00013367564,0.00032464648,0.023834398,0.0018112368],"genre_scores_gemma":[0.7443845,0.00017515762,0.25375837,0.000045969966,0.000014202463,0.00008809839,0.00039383827,0.00045623493,0.00068350614],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99839383,0.00032466353,0.00010391269,0.0003301073,0.00068944565,0.00015801439],"domain_scores_gemma":[0.9937709,0.003611889,0.0009670563,0.0008109269,0.0006833862,0.00015574711],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00073351763,0.0013418085,0.0008331486,0.003515103,0.00040388387,0.0008597788,0.0012376208,0.0007852932,0.00164526],"category_scores_gemma":[0.00794435,0.00041793642,0.0006967172,0.001363767,0.0015007163,0.0014275659,0.0011772116,0.00074421294,0.00034498618],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071982277,0.00033772326,0.03985266,0.0009077449,0.00019214906,0.0015674467,0.0011779715,0.21167749,0.15630525,0.0114527205,0.0030872307,0.5727217],"study_design_scores_gemma":[0.000028619928,0.00010658173,0.0025273673,0.000037007776,0.000048095753,0.00037637466,0.00005524775,0.95301664,0.037674304,0.005070136,0.0010260453,0.00003355138],"about_ca_topic_score_codex":0.0036856532,"about_ca_topic_score_gemma":0.0032117178,"teacher_disagreement_score":0.0036856532,"about_ca_system_score_codex":0.00067375344,"about_ca_system_score_gemma":0.0012911353,"threshold_uncertainty_score":0.007328391},"labels":[],"label_agreement":null},{"id":"W2593244908","doi":"10.1145/2700529","title":"Automated Bug Finding in Video Games","year":2017,"lang":"en","type":"article","venue":"Computers in entertainment","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Video game; Video game development; Game development tool; Video game design; Game Developer; Instrumentation (computer programming); Process (computing); Sequential game; Game programming; State (computer science); Sample (material); Game testing; Game design; Event (particle physics); Programming language; Human–computer interaction; Game design document; Game theory; Game art design; Multimedia","score_opus":0.025209883546195373,"score_gpt":0.3043409565355983,"score_spread":0.27913107298940293,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2593244908","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8448062,0.00040709594,0.14536437,0.00013647377,0.000020797572,0.00017222574,0.00028331144,0.0076950146,0.0011144907],"genre_scores_gemma":[0.9245403,0.00008293224,0.07445372,0.000033516953,0.0000047258977,0.00005123482,0.00027247547,0.00018027262,0.00038088977],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99628663,0.0013449386,0.000233768,0.00088076683,0.0009780611,0.00027589084],"domain_scores_gemma":[0.989179,0.0059843645,0.0019343351,0.001606399,0.0010168582,0.00027901537],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017952619,0.00072859845,0.00067220547,0.0015653626,0.00045238764,0.0009817306,0.0016349124,0.0006621558,0.0006100714],"category_scores_gemma":[0.0159086,0.00062073465,0.0004262629,0.0007839227,0.00088680524,0.0011975238,0.0010750557,0.00069635944,0.00022210907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014767154,0.0010551953,0.13890678,0.0006710538,0.0002618995,0.0016810737,0.004148748,0.065665044,0.17623022,0.0067471536,0.0034112935,0.5997448],"study_design_scores_gemma":[0.00017514176,0.0010201485,0.084756255,0.00009978972,0.00015633894,0.0014273819,0.0007630285,0.7711661,0.1262508,0.008486078,0.005583743,0.00011525293],"about_ca_topic_score_codex":0.006172969,"about_ca_topic_score_gemma":0.0066115553,"teacher_disagreement_score":0.006172969,"about_ca_system_score_codex":0.0007456793,"about_ca_system_score_gemma":0.0008277885,"threshold_uncertainty_score":0.012274027},"labels":[],"label_agreement":null},{"id":"W2598381280","doi":"10.1145/3066862.3066867","title":"Fun and games at IEEE WCCI 2016, Vancouver, Canada","year":2017,"lang":"en","type":"article","venue":"ACM SIGEVOlution","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council","keywords":"Champion; Competition (biology); Artificial intelligence; Media studies; Psychology; Operations research; Computer science; Sociology; Engineering; Political science; Law","score_opus":0.017929992714591437,"score_gpt":0.24133561414956178,"score_spread":0.22340562143497034,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2598381280","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00873118,0.015962824,0.019910377,0.014362508,0.0107426075,0.00017000684,0.005450582,0.0041069007,0.920563],"genre_scores_gemma":[0.009897244,0.003397318,0.0038821355,0.00025561344,0.00035580658,0.000039716055,0.0017981377,0.00041716712,0.97995687],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99957246,0.00003909002,0.00001283024,0.00009644886,0.00018186634,0.00009735477],"domain_scores_gemma":[0.9989672,0.000060598035,0.000015167999,0.000067705165,0.00035082176,0.00053861784],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0008245116,0.0014357618,0.0005825326,0.0011545934,0.0020856853,0.0036088836,0.0009547326,0.0009794426,0.29992],"category_scores_gemma":[0.0014026712,0.00034857617,0.00029904814,0.0012501308,0.0006216902,0.0015317621,0.0016181198,0.0015390554,0.1070135],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007802778,0.000065878456,0.00059428776,0.00008238036,0.0000074209315,0.000082344464,0.000114897135,0.0005167251,0.0011458878,0.0057510817,0.86495835,0.12660277],"study_design_scores_gemma":[0.00000908625,0.000018573925,0.0011725135,0.000080442886,0.000004492905,0.00004596994,0.00014876373,0.0011429059,0.0005118501,0.0029342219,0.9939201,0.000010950978],"about_ca_topic_score_codex":0.071283385,"about_ca_topic_score_gemma":0.23822106,"teacher_disagreement_score":0.29992,"about_ca_system_score_codex":0.0034456814,"about_ca_system_score_gemma":0.0037617255,"threshold_uncertainty_score":0.99857914},"labels":[],"label_agreement":null},{"id":"W2598672795","doi":"10.1007/978-3-662-54494-5_2","title":"Bordeaux: A Tool for Thinking Outside the Box","year":2017,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Negation; Extension (predicate logic); Meaning (existential); Programming language; Algorithm; Theoretical computer science","score_opus":0.02714208637281054,"score_gpt":0.28431637315416347,"score_spread":0.25717428678135296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2598672795","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026416786,0.0022331004,0.7622902,0.0019164965,0.0013490843,0.000097306925,0.0008077193,0.078929596,0.14973477],"genre_scores_gemma":[0.059244193,0.0020435352,0.6856302,0.0012723939,0.00042371103,0.00026029974,0.001999854,0.036759052,0.21236676],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99906725,0.00027693488,0.000047156143,0.00017982093,0.00032444563,0.000104458966],"domain_scores_gemma":[0.99884254,0.00063123327,0.000038893566,0.00017445274,0.0001934704,0.00011937841],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001177694,0.0014641316,0.0007626488,0.0020809537,0.0012738962,0.0054395874,0.0017472483,0.0017244772,0.08172629],"category_scores_gemma":[0.0035816738,0.0011596539,0.0014234667,0.000996461,0.001388504,0.0054334947,0.0024733678,0.003013876,0.028938867],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028376988,0.000057902445,0.00037508138,0.000479057,0.000036850128,0.00040064825,0.0016023722,0.0023419703,0.0067570363,0.4127224,0.23551275,0.3394301],"study_design_scores_gemma":[0.000036556947,0.00002544766,0.00021462426,0.00021671911,0.000026750991,0.0004672417,0.00017470874,0.00734943,0.0044215797,0.07230329,0.91469777,0.00006601043],"about_ca_topic_score_codex":0.0035261803,"about_ca_topic_score_gemma":0.0049212985,"teacher_disagreement_score":0.08172629,"about_ca_system_score_codex":0.0011775251,"about_ca_system_score_gemma":0.0010406332,"threshold_uncertainty_score":0.27340168},"labels":[],"label_agreement":null},{"id":"W2598705408","doi":"10.1007/978-3-319-55792-2_4","title":"Focusing Learning-Based Testing Away from Known Weaknesses","year":2017,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Strengths and weaknesses; Similarity (geometry); Artificial intelligence; Adversary; Machine learning; Extension (predicate logic); Test (biology); Measure (data warehouse); Similarity measure; Theoretical computer science; Data mining; Computer security; Programming language","score_opus":0.03557945597157287,"score_gpt":0.27079144967597146,"score_spread":0.2352119937043986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2598705408","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025187125,0.0011015331,0.9374942,0.0008936792,0.00013398536,0.00011178104,0.00008097271,0.0053757466,0.029620873],"genre_scores_gemma":[0.42484316,0.0011017419,0.5446817,0.0008619965,0.00018281132,0.00010169279,0.00050895906,0.00173571,0.025982216],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99766195,0.0005434603,0.00011151224,0.00046091853,0.0010380271,0.00018413477],"domain_scores_gemma":[0.9848564,0.009171662,0.0008090008,0.0028453916,0.001923031,0.0003944139],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023055344,0.0017512427,0.0008582252,0.0014135817,0.00038705487,0.0017674329,0.0029703039,0.0012397977,0.008928966],"category_scores_gemma":[0.014353725,0.00050070893,0.0008147601,0.00094113755,0.0009666889,0.0046131816,0.0028068668,0.0031109198,0.0033735684],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001338403,0.00027574433,0.0028786857,0.00040338983,0.00005184948,0.00020146088,0.0004263912,0.0153774535,0.019267093,0.02457377,0.0073154755,0.92909485],"study_design_scores_gemma":[0.00010571138,0.0007721375,0.006927392,0.0010832053,0.0003563118,0.002823532,0.00069291884,0.43442664,0.11480739,0.37472492,0.06314933,0.00013045473],"about_ca_topic_score_codex":0.0005854409,"about_ca_topic_score_gemma":0.0012507876,"teacher_disagreement_score":0.008928966,"about_ca_system_score_codex":0.00063700357,"about_ca_system_score_gemma":0.0010960584,"threshold_uncertainty_score":0.029870331},"labels":[],"label_agreement":null},{"id":"W2607146905","doi":"10.1002/smr.1868","title":"Extending Category Partition's <scp>B</scp>ase <scp>C</scp>hoice criterion to better support constraints","year":2017,"lang":"en","type":"article","venue":"Journal of Software Evolution and Process","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Partition (number theory); Base (topology); Computer science; Set (abstract data type); Mathematical optimization; Mathematics; Combinatorics; Programming language","score_opus":0.025064019517953475,"score_gpt":0.30153339368785353,"score_spread":0.2764693741699001,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2607146905","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12389311,0.0002865722,0.8533914,0.00080199464,0.00008621975,0.0006251775,0.00041769163,0.00089407986,0.019603705],"genre_scores_gemma":[0.76047623,0.000067501525,0.23590107,0.0003313479,0.000053681077,0.00035340353,0.00041685716,0.00021365497,0.002186255],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9828477,0.0051702806,0.00093824085,0.0015053105,0.008366917,0.0011715413],"domain_scores_gemma":[0.9320783,0.040006842,0.0033800898,0.0065157535,0.015648564,0.0023705591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00985782,0.0010663774,0.0013462983,0.0048131566,0.0013336036,0.0025591992,0.0023646678,0.0022455724,0.0060083377],"category_scores_gemma":[0.04610426,0.00052579254,0.0015013875,0.002430366,0.0035241018,0.0050232657,0.0042106407,0.002138077,0.00079289003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013840828,0.0007531507,0.034630798,0.0007077167,0.00034807538,0.001602867,0.0012880793,0.2709143,0.024870818,0.3867269,0.011629308,0.2651439],"study_design_scores_gemma":[0.0001471537,0.0008936247,0.009744897,0.0002762177,0.00011120846,0.0013182473,0.0004953038,0.7289489,0.018579602,0.22400115,0.015293964,0.00018982362],"about_ca_topic_score_codex":0.0056943945,"about_ca_topic_score_gemma":0.0054517305,"teacher_disagreement_score":0.00985782,"about_ca_system_score_codex":0.001871784,"about_ca_system_score_gemma":0.0029430843,"threshold_uncertainty_score":0.05213374},"labels":[],"label_agreement":null},{"id":"W2607355777","doi":"","title":"Requirement-based Software Testing With the UML: A Systematic Mapping Study","year":2012,"lang":"en","type":"article","venue":"International Conference on Software Engineering Advances","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Software testing; Unified Modeling Language; Software engineering; Software performance testing; Applications of UML; Software; Reliability engineering; Programming language; Software development; Software construction; Engineering","score_opus":0.0600898824269571,"score_gpt":0.2875042514738007,"score_spread":0.22741436904684362,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2607355777","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8457267,0.0034995072,0.13919829,0.00037674227,0.000027537119,0.0032982838,0.0004830397,0.00027593892,0.007114],"genre_scores_gemma":[0.9124456,0.0014360683,0.083625905,0.00013073419,0.0000063747357,0.0010583072,0.00042033495,0.00007051954,0.0008061081],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.9688965,0.021369573,0.002843849,0.0014714153,0.0048683197,0.000550334],"domain_scores_gemma":[0.81745315,0.13899651,0.010888449,0.010894022,0.021108111,0.0006597219],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.023104073,0.0007296174,0.00071653345,0.008576163,0.0013729072,0.0014229111,0.0014666603,0.00086656417,0.0012018615],"category_scores_gemma":[0.10643105,0.00064400624,0.0011342069,0.0050317263,0.0012393898,0.0034538354,0.003070647,0.00095491234,0.00022813355],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005327838,0.0024276297,0.14250576,0.007871989,0.0005697935,0.0016605264,0.11739616,0.0064760847,0.013834295,0.013050514,0.0014702709,0.6922042],"study_design_scores_gemma":[0.0008229186,0.01004312,0.41641873,0.02367647,0.00545905,0.015269672,0.19061375,0.10851704,0.10393166,0.039592456,0.08499989,0.0006552622],"about_ca_topic_score_codex":0.0044905776,"about_ca_topic_score_gemma":0.007983798,"teacher_disagreement_score":0.99142385,"about_ca_system_score_codex":0.0021981741,"about_ca_system_score_gemma":0.0067897164,"threshold_uncertainty_score":0.122187495},"labels":[],"label_agreement":null},{"id":"W2611833875","doi":"","title":"Verification of Modular Systems with Unknown Components Combining Testing and Inference","year":2012,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Modular design; Computer science; Component (thermodynamics); TRACE (psycholinguistics); Reachability; Construct (python library); Model checking; Model-based testing; Isolation (microbiology); Inference; Theoretical computer science; Algorithm; Test case; Programming language; Artificial intelligence; Machine learning","score_opus":0.03582087525383267,"score_gpt":0.2392506539911267,"score_spread":0.203429778737294,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2611833875","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057234652,0.00041219915,0.93806356,0.0003870691,0.00011361517,0.000054184336,0.00007615921,0.0020611794,0.0015974063],"genre_scores_gemma":[0.6592849,0.00035324084,0.33681566,0.00017522743,0.0002270301,0.000067556,0.00028353362,0.0006230766,0.0021697325],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99302614,0.0025377534,0.0005240358,0.0012748797,0.0020805518,0.00055655575],"domain_scores_gemma":[0.97346514,0.020342734,0.0008445741,0.003821659,0.0012438297,0.00028205346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006140932,0.0017834498,0.002238903,0.0015877314,0.00072930247,0.0024835167,0.0025741248,0.0017606247,0.0028779868],"category_scores_gemma":[0.020327872,0.001857836,0.003732761,0.0010527129,0.0039414307,0.0049942452,0.0035222133,0.0027541416,0.00048729964],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013342336,0.0005309913,0.0091334125,0.0010321669,0.0009495559,0.0019068958,0.00086883554,0.3597834,0.05388352,0.2404267,0.0050497623,0.32510045],"study_design_scores_gemma":[0.00013821329,0.00014517806,0.0013538685,0.000072858784,0.0002720018,0.00021351724,0.000037745183,0.68191457,0.029824223,0.28336808,0.0026069563,0.00005283783],"about_ca_topic_score_codex":0.0017796507,"about_ca_topic_score_gemma":0.0022176502,"teacher_disagreement_score":0.006140932,"about_ca_system_score_codex":0.0013499613,"about_ca_system_score_gemma":0.0016892947,"threshold_uncertainty_score":0.032476723},"labels":[],"label_agreement":null},{"id":"W2615727104","doi":"","title":"Successful Implementation of Protocol V","year":2006,"lang":"en","type":"article","venue":"JMU Scholoraly Commons (James Madison University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Mine action; Protocol (science); Computer science; Computer network; Political science; Medicine; Law","score_opus":0.012324386388021514,"score_gpt":0.2732457354490791,"score_spread":0.2609213490610576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2615727104","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015351129,0.00029779508,0.6746412,0.022810332,0.009010519,0.027985565,0.0036715174,0.015191011,0.23104092],"genre_scores_gemma":[0.37481645,0.00070443784,0.28909236,0.024436606,0.002196833,0.051393233,0.006540169,0.0043893796,0.24643056],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9060458,0.038349647,0.00651284,0.00505795,0.03486271,0.009170966],"domain_scores_gemma":[0.876759,0.04252426,0.0037434963,0.04390834,0.030394817,0.002670127],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.054972157,0.00083687506,0.0012637471,0.0017356813,0.002939799,0.008979711,0.0026583741,0.006294037,0.042431526],"category_scores_gemma":[0.16943121,0.0013202135,0.0014767676,0.0010709076,0.002942056,0.0053030793,0.0057746833,0.006617966,0.030715441],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010637504,0.00039543066,0.0016184548,0.0007381483,0.0001813683,0.0013027072,0.0046193223,0.0019452088,0.011387298,0.4683617,0.3626003,0.14578636],"study_design_scores_gemma":[0.000639436,0.0007152993,0.0019673987,0.0008086159,0.00012216467,0.0008622983,0.0016151232,0.010114411,0.037024178,0.12578347,0.82005924,0.00028833834],"about_ca_topic_score_codex":0.002537355,"about_ca_topic_score_gemma":0.0026055102,"teacher_disagreement_score":0.054972157,"about_ca_system_score_codex":0.0031835667,"about_ca_system_score_gemma":0.017626008,"threshold_uncertainty_score":0.29072404},"labels":[],"label_agreement":null},{"id":"W2620742065","doi":"10.1109/icse-c.2017.6","title":"JSDeodorant: Class-Awareness for JavaScript Programs","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Namespace; JavaScript; Computer science; Programming language; Inheritance (genetic algorithm); Callback; Class (philosophy); Modular design; Eclipse; Unobtrusive JavaScript; Software engineering; Object-oriented programming; Object (grammar); Common Object Request Broker Architecture; World Wide Web; Operating system; Artificial intelligence; Rich Internet application","score_opus":0.09250601432181466,"score_gpt":0.3382746358746273,"score_spread":0.24576862155281265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2620742065","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011148319,0.00053392537,0.7581577,0.00028565223,0.00022905077,0.00027318863,0.00086356007,0.22072022,0.0077884537],"genre_scores_gemma":[0.19934435,0.001277438,0.6661959,0.001151838,0.00028128907,0.00082426716,0.0069106403,0.10168987,0.022324381],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982084,0.00023646849,0.00019243664,0.00041093415,0.0007921484,0.00015967424],"domain_scores_gemma":[0.9951198,0.0021276758,0.00037691157,0.0014775471,0.0006055386,0.0002924646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002576299,0.0012126438,0.0007538296,0.0012126201,0.0005435686,0.002240833,0.002698357,0.0012997233,0.0048902114],"category_scores_gemma":[0.010484391,0.0013539967,0.0010737308,0.0005278506,0.0010039327,0.005791391,0.003158989,0.0028346775,0.0032132387],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011912485,0.00067961565,0.010336248,0.0021430026,0.00022019309,0.001806062,0.0052920585,0.00336435,0.1329125,0.044842333,0.1335806,0.66363174],"study_design_scores_gemma":[0.00032410762,0.00030412877,0.008110863,0.0007186727,0.00023657377,0.0037694885,0.00030066154,0.08632324,0.22011314,0.035437558,0.6440265,0.00033516373],"about_ca_topic_score_codex":0.0009680829,"about_ca_topic_score_gemma":0.0018332378,"teacher_disagreement_score":0.0048902114,"about_ca_system_score_codex":0.0004751617,"about_ca_system_score_gemma":0.001008936,"threshold_uncertainty_score":0.016359389},"labels":[],"label_agreement":null},{"id":"W270596444","doi":"10.1007/978-3-319-17040-4_14","title":"A Formal Approach to Verify Completeness and Detect Anomalies in Firewall Security Policies","year":2015,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Security policy; Automaton; Firewall (physics); Completeness (order theory); Computer security model; Computer security; Formalism (music); Formal methods; Theoretical computer science; Software engineering; Spacetime","score_opus":0.03820750955614092,"score_gpt":0.26836447362602334,"score_spread":0.23015696406988242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W270596444","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018230856,0.00004245626,0.9956393,0.00020913135,0.00003935867,0.00012106029,0.000057628666,0.0011127637,0.0009550928],"genre_scores_gemma":[0.109946914,0.00016972072,0.884823,0.00036374282,0.00013110337,0.00045250636,0.000299085,0.00055307103,0.0032609212],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.986816,0.0036190513,0.0015651456,0.0016377756,0.0051540695,0.0012078936],"domain_scores_gemma":[0.9597415,0.024831075,0.0017727576,0.00805378,0.0049948,0.0006060634],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009960968,0.0021525545,0.0015524013,0.002995676,0.0021517323,0.0044679707,0.005604009,0.0030653032,0.00585617],"category_scores_gemma":[0.02915643,0.0034450004,0.005560823,0.0015435911,0.008916153,0.012663095,0.006888776,0.007834659,0.0015369005],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018145132,0.00047909908,0.0014860465,0.0007153654,0.00020198524,0.00054367556,0.0011012204,0.07691816,0.015079104,0.792514,0.0053650416,0.105414815],"study_design_scores_gemma":[0.00013629413,0.00014185626,0.0002452974,0.00020800813,0.00020852534,0.00037481586,0.00019848841,0.362492,0.022617707,0.6014644,0.011806038,0.00010654963],"about_ca_topic_score_codex":0.005216467,"about_ca_topic_score_gemma":0.0063016172,"teacher_disagreement_score":0.009960968,"about_ca_system_score_codex":0.002690404,"about_ca_system_score_gemma":0.00594241,"threshold_uncertainty_score":0.05267924},"labels":[],"label_agreement":null},{"id":"W2733241474","doi":"10.5555/3105427.3105430","title":"Searching for behavioural bugs with stateful test oracles in web crawlers","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Stateful firewall; Computer science; Crawling; Web crawler; World Wide Web; Web application; Software engineering; Computer security","score_opus":0.0564857498765327,"score_gpt":0.3196836482819782,"score_spread":0.2631978984054455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2733241474","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55056083,0.0006056789,0.41304407,0.0006573178,0.00006193409,0.00025259992,0.0005291205,0.032315623,0.0019727952],"genre_scores_gemma":[0.84845805,0.0001223428,0.14926445,0.00013135008,0.00002031221,0.0001153061,0.00067571044,0.0005911134,0.00062130444],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9931972,0.002934407,0.0005531381,0.00093068636,0.0018704596,0.000514078],"domain_scores_gemma":[0.9600666,0.026632257,0.0051179132,0.0045679393,0.0027025817,0.000912696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00417831,0.0010323029,0.0009943286,0.003180235,0.0005489209,0.0020946108,0.0018162645,0.001703016,0.0008812438],"category_scores_gemma":[0.042289853,0.0011149899,0.000825438,0.0013968094,0.001444405,0.0032823856,0.0016180013,0.0012582097,0.000395466],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014405685,0.0018312992,0.20243457,0.000910058,0.00032965885,0.0043496503,0.0032285268,0.13620187,0.06344958,0.021794258,0.008353448,0.55567664],"study_design_scores_gemma":[0.00009507017,0.0004049839,0.0125204185,0.00011565301,0.00010302139,0.0011327842,0.00032728078,0.9350196,0.030096984,0.018075434,0.0020293572,0.00007927397],"about_ca_topic_score_codex":0.0030284626,"about_ca_topic_score_gemma":0.0040078997,"teacher_disagreement_score":0.00417831,"about_ca_system_score_codex":0.0006564413,"about_ca_system_score_gemma":0.0011986558,"threshold_uncertainty_score":0.02209729},"labels":[],"label_agreement":null},{"id":"W2733392061","doi":"10.5555/3105427.3105436","title":"JTeXpert at the SBST 2017 tool competition","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Test suite; Unit testing; Suite; CONTEST; Computer science; Java; Programming language; Software; Code (set theory); Competition (biology); Software testing; Source code; Test (biology); Test case; Artificial intelligence; Software engineering; Set (abstract data type); Machine learning","score_opus":0.040837899822264774,"score_gpt":0.29927517210164556,"score_spread":0.2584372722793808,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2733392061","genre_codex":"empirical","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.626713,0.009933814,0.13719968,0.008264362,0.009689054,0.0019929984,0.024694625,0.077344626,0.104167804],"genre_scores_gemma":[0.6840777,0.0013300396,0.108784385,0.0017944826,0.00085539836,0.0010326111,0.12882553,0.011046956,0.062252928],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9767129,0.005644026,0.0008692814,0.0022638205,0.011107173,0.0034028825],"domain_scores_gemma":[0.9646145,0.00993275,0.00072407787,0.002854687,0.014152401,0.0077215303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017770026,0.0030540598,0.0021506944,0.007266471,0.0017775054,0.0037857725,0.0028924043,0.002996385,0.015717128],"category_scores_gemma":[0.031644948,0.00078100356,0.0021932903,0.0044601364,0.0012333086,0.0037150807,0.0044329823,0.0034652394,0.009710255],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033756525,0.0034054765,0.011655196,0.0010126029,0.00046484504,0.0010032278,0.0009871892,0.015735723,0.01781773,0.009338519,0.5468537,0.38835007],"study_design_scores_gemma":[0.0046359655,0.009266671,0.058263883,0.00061457785,0.00041022562,0.0018889329,0.0013496358,0.186908,0.05856916,0.015047087,0.66235113,0.0006947588],"about_ca_topic_score_codex":0.010264193,"about_ca_topic_score_gemma":0.019920213,"teacher_disagreement_score":0.017770026,"about_ca_system_score_codex":0.0030685791,"about_ca_system_score_gemma":0.0031033622,"threshold_uncertainty_score":0.09397799},"labels":[],"label_agreement":null},{"id":"W2733860488","doi":"","title":"Expected time to detection of interaction faults","year":2013,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Mathematics; Econometrics; Statistics","score_opus":0.012411871599337737,"score_gpt":0.2533188563413243,"score_spread":0.24090698474198655,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2733860488","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.71373963,0.0014905373,0.25898126,0.005260828,0.0008468404,0.0005320565,0.0028739776,0.0042977203,0.011977154],"genre_scores_gemma":[0.9696186,0.0002660445,0.024304757,0.00047031845,0.0002247488,0.00013296992,0.0014570968,0.00031141684,0.003214137],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9932295,0.0012050974,0.00046247503,0.0012286789,0.002180217,0.0016939695],"domain_scores_gemma":[0.87210774,0.10314259,0.0059040263,0.004008255,0.008879202,0.005958231],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005327928,0.0011285971,0.0015951929,0.0018533054,0.0010294577,0.0027914084,0.0026178828,0.0021871065,0.010612017],"category_scores_gemma":[0.07312012,0.0009445318,0.0015025631,0.0008101043,0.0012223788,0.0030056743,0.0016762712,0.0024109662,0.0008704603],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.016316473,0.0020802214,0.08105061,0.0020322734,0.001250263,0.0025897822,0.00077159866,0.53820246,0.042527672,0.092609204,0.016802324,0.20376706],"study_design_scores_gemma":[0.00086144335,0.0021273918,0.016135655,0.00014659327,0.00050961744,0.0021369276,0.0002995222,0.8657018,0.022144737,0.087771334,0.0020418111,0.00012320586],"about_ca_topic_score_codex":0.002574578,"about_ca_topic_score_gemma":0.002591873,"teacher_disagreement_score":0.010612017,"about_ca_system_score_codex":0.0022967677,"about_ca_system_score_gemma":0.0051485645,"threshold_uncertainty_score":0.035500705},"labels":[],"label_agreement":null},{"id":"W2736040703","doi":"10.1145/3092703.3098229","title":"SealTest: a simple library for test sequence generation","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Simple (philosophy); Computer science; Sequence (biology); Test (biology); Chemistry","score_opus":0.11375339780369222,"score_gpt":0.3372430159476951,"score_spread":0.22348961814400292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2736040703","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021248076,0.00015084633,0.7756823,0.000059934577,0.00006748195,0.0003168996,0.0038029458,0.21445799,0.0033368357],"genre_scores_gemma":[0.08620789,0.0006020195,0.76820105,0.00064815686,0.00013961515,0.002636726,0.035824712,0.08879555,0.016944261],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9971078,0.00058659964,0.0004051401,0.00042579562,0.0012392299,0.00023548599],"domain_scores_gemma":[0.99255776,0.0040859645,0.00065307633,0.0012698448,0.0012315004,0.00020192596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027070004,0.0022216644,0.0012035576,0.0029084522,0.0005897811,0.0017099166,0.0033972904,0.0014695541,0.046194337],"category_scores_gemma":[0.014132029,0.001610519,0.0018181846,0.0017190138,0.00090078946,0.002783943,0.0020879356,0.0018210479,0.019233773],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013825153,0.0005976258,0.006605214,0.0036680764,0.00039183907,0.0012516367,0.00057304156,0.040143013,0.06095722,0.026491122,0.23570284,0.62223583],"study_design_scores_gemma":[0.0010244254,0.0008812052,0.0041405507,0.0006649853,0.0002749317,0.002970564,0.00015112515,0.39871734,0.22002448,0.061326154,0.30929592,0.0005284365],"about_ca_topic_score_codex":0.0023172165,"about_ca_topic_score_gemma":0.0028436156,"teacher_disagreement_score":0.046194337,"about_ca_system_score_codex":0.0008052621,"about_ca_system_score_gemma":0.0015447675,"threshold_uncertainty_score":0.15453547},"labels":[],"label_agreement":null},{"id":"W2740565296","doi":"10.1145/3106237.3106258","title":"QTEP: quality-aware test case prioritization","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code coverage; Computer science; Source code; Software quality; Leverage (statistics); Test case; Test suite; Code (set theory); Code review; Reliability engineering; Software; Programming language; Software development; Set (abstract data type); Engineering","score_opus":0.05778218964391741,"score_gpt":0.36462649522235946,"score_spread":0.30684430557844206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740565296","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016974078,0.0011011094,0.95370966,0.00053915277,0.00021988673,0.00096949254,0.0008476953,0.021989435,0.0036495135],"genre_scores_gemma":[0.28686726,0.00065842323,0.6984928,0.0006496446,0.00028369558,0.0009461734,0.0048628487,0.0033426525,0.0038965521],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9914869,0.002270693,0.00085425493,0.0011305708,0.003637558,0.000619944],"domain_scores_gemma":[0.9736462,0.012779175,0.0021108738,0.0047776164,0.005727972,0.00095807813],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00737351,0.0021469726,0.0012278465,0.00436681,0.00069778407,0.0021841573,0.0034060115,0.0011138173,0.007585159],"category_scores_gemma":[0.033324026,0.0009681184,0.0014345823,0.001899832,0.00089345453,0.0030945484,0.0032775952,0.0023840386,0.0019382603],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012289245,0.00061630114,0.012174837,0.0015427975,0.000368818,0.0007206638,0.00045422444,0.055973813,0.046037547,0.019607967,0.035862543,0.8254116],"study_design_scores_gemma":[0.00076994987,0.0008670585,0.006409974,0.00034138796,0.0004075459,0.0014241656,0.0002870927,0.84294784,0.061074167,0.04180416,0.043524973,0.00014168562],"about_ca_topic_score_codex":0.003547668,"about_ca_topic_score_gemma":0.004457746,"teacher_disagreement_score":0.007585159,"about_ca_system_score_codex":0.0011900268,"about_ca_system_score_gemma":0.0035589647,"threshold_uncertainty_score":0.038995326},"labels":[],"label_agreement":null},{"id":"W2741328617","doi":"10.1145/3106237.3106274","title":"Better test cases for better automated program repair","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":126,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Overfitting; Metric (unit); Artificial intelligence; Benchmark (surveying); Test (biology); Machine learning; Software bug; Pattern recognition (psychology); Data mining; Software; Programming language; Engineering","score_opus":0.044205879285435255,"score_gpt":0.34301408199295436,"score_spread":0.2988082027075191,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2741328617","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19926848,0.0015958387,0.74176794,0.0019456627,0.00022249181,0.00072111137,0.0014540064,0.047635626,0.0053887824],"genre_scores_gemma":[0.46634698,0.0004913442,0.52334785,0.0008011628,0.00009885047,0.00031503418,0.004065966,0.0031358865,0.0013969741],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98543984,0.0059185983,0.0014028011,0.0016565017,0.004851455,0.0007308448],"domain_scores_gemma":[0.92143583,0.04815075,0.0074306824,0.014127582,0.007956191,0.00089894154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008221186,0.0026272133,0.0012383325,0.0061994996,0.00042589984,0.0026842018,0.0026322147,0.0019159346,0.0066808285],"category_scores_gemma":[0.06377773,0.0010286816,0.0013896987,0.0020045615,0.0011508233,0.004943328,0.0020408458,0.0021686978,0.0022717512],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008067993,0.0019192699,0.05632962,0.0013477211,0.00023678652,0.0026895208,0.00082359044,0.12320418,0.08896373,0.016658226,0.025157258,0.68186337],"study_design_scores_gemma":[0.00053670816,0.001064177,0.0140745,0.0007117449,0.0002586574,0.0022011811,0.0004753996,0.80564183,0.11734995,0.018958481,0.038546737,0.00018052342],"about_ca_topic_score_codex":0.0028717346,"about_ca_topic_score_gemma":0.0033495796,"teacher_disagreement_score":0.008221186,"about_ca_system_score_codex":0.0010945122,"about_ca_system_score_gemma":0.0018655732,"threshold_uncertainty_score":0.04347825},"labels":[],"label_agreement":null},{"id":"W2747184587","doi":"10.1016/bs.adcom.2017.06.002","title":"Testing the Control-Flow, Data-Flow, and Time Aspects of Communication Systems: A Survey","year":2017,"lang":"en","type":"book-chapter","venue":"Advances in computers","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke; Concordia University","funders":"","keywords":"Extended finite-state machine; Computer science; Conformance testing; Finite-state machine; Standardization; Data flow diagram; Software; Deterministic finite automaton; Database; Programming language; Operating system","score_opus":0.05862950462809106,"score_gpt":0.29601633582158937,"score_spread":0.23738683119349832,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2747184587","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011964066,0.7956653,0.13235556,0.002345202,0.0008827388,0.0001068418,0.00032457671,0.0010060428,0.055349715],"genre_scores_gemma":[0.07505451,0.8088853,0.083230406,0.0013126935,0.0015010271,0.00014003419,0.0010567639,0.00033967532,0.028479643],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99869776,0.00018108997,0.00009808241,0.00016934826,0.0007845332,0.000069130736],"domain_scores_gemma":[0.995411,0.003514654,0.0001549545,0.00018552081,0.00062459504,0.000109224435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010217436,0.0010929337,0.0009399994,0.0030897153,0.00032197172,0.001707682,0.001241183,0.00086837437,0.0041031726],"category_scores_gemma":[0.004019442,0.0005324,0.00045726186,0.005584269,0.001071227,0.0036740415,0.00077311334,0.0014523525,0.0014508147],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026974449,0.00008722444,0.0012225134,0.0014996672,0.000017066897,0.000073326744,0.00014747192,0.002322204,0.0017716801,0.020716803,0.01498362,0.9571313],"study_design_scores_gemma":[0.00004104032,0.00056181266,0.009041495,0.005093603,0.00018058586,0.0043991394,0.00072272454,0.025135694,0.013798736,0.1906866,0.750234,0.00010454648],"about_ca_topic_score_codex":0.0018991197,"about_ca_topic_score_gemma":0.0025311788,"teacher_disagreement_score":0.0041031726,"about_ca_system_score_codex":0.0007174908,"about_ca_system_score_gemma":0.0012550714,"threshold_uncertainty_score":0.013726532},"labels":[],"label_agreement":null},{"id":"W2754010186","doi":"10.1109/compsac.2017.138","title":"How Does GUI Testing Exercise Application Logic Functionality?","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Programming language; Graphical user interface; Graphical user interface testing; Keyword-driven testing; Regression testing; Code coverage; Software; White-box testing; Software engineering; User interface; Software system; Software construction; User interface design","score_opus":0.05614546730983575,"score_gpt":0.28686183270118853,"score_spread":0.2307163653913528,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2754010186","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49039102,0.0020535614,0.4446117,0.010679102,0.00023149521,0.00020952248,0.000098883866,0.0026976713,0.049027003],"genre_scores_gemma":[0.9551311,0.00039711213,0.042109903,0.0010490582,0.00006177472,0.000055705386,0.000058873997,0.000250842,0.00088547386],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9734997,0.015419461,0.000860147,0.0018858629,0.0071605574,0.001174273],"domain_scores_gemma":[0.8129095,0.15207039,0.009498771,0.013171764,0.010873765,0.0014757887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02037416,0.0008835033,0.00061804504,0.0015829426,0.0005955767,0.0032466548,0.0017498095,0.002567944,0.0021826],"category_scores_gemma":[0.16859497,0.0004424871,0.0005694252,0.00096187164,0.00387129,0.010845767,0.0013030904,0.0014898847,0.0011420741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058760523,0.0010472798,0.13046704,0.00072982826,0.0002441855,0.0006367369,0.0045354306,0.016049847,0.023178937,0.06293746,0.0034665684,0.7561191],"study_design_scores_gemma":[0.00047191413,0.0071713794,0.2550475,0.0020833975,0.0006625792,0.008566499,0.009834236,0.16348524,0.1347862,0.36543766,0.05193442,0.00051901495],"about_ca_topic_score_codex":0.0021296102,"about_ca_topic_score_gemma":0.0020130887,"teacher_disagreement_score":0.02037416,"about_ca_system_score_codex":0.0008384335,"about_ca_system_score_gemma":0.001102336,"threshold_uncertainty_score":0.10775018},"labels":[],"label_agreement":null},{"id":"W2755609451","doi":"10.1109/compsac.2017.221","title":"State-Based Tests Suites Automatic Generation Tool (STAGE-1)","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Test suite; Tree traversal; Graph traversal; Automation; Test case; Random testing; Code coverage; Model-based testing; Test Management Approach; Graph; Test (biology); Finite-state machine; State (computer science); Keyword-driven testing; Software; Theoretical computer science; Programming language; Machine learning; Software system; Software construction; Engineering","score_opus":0.04925087200966932,"score_gpt":0.3073691148616308,"score_spread":0.2581182428519615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2755609451","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013415816,0.000060624567,0.92808205,0.00006633271,0.000032785694,0.00038365083,0.0011986517,0.053796247,0.0029637956],"genre_scores_gemma":[0.18654656,0.00012667665,0.8001827,0.000083201216,0.000022724915,0.0010343279,0.005286756,0.0037533606,0.0029637492],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99839824,0.0005148636,0.00015666493,0.00025636717,0.0005553751,0.0001185255],"domain_scores_gemma":[0.99440634,0.003532756,0.00031188803,0.0007725916,0.0008747507,0.00010166712],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017177034,0.0010714261,0.00052049593,0.0025089667,0.00035531702,0.0012210121,0.0011091843,0.00073961174,0.009864866],"category_scores_gemma":[0.010648818,0.0005280169,0.001048245,0.0009613022,0.0004856488,0.001124142,0.00086407334,0.0008357056,0.0027446407],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070234796,0.0005283634,0.009449714,0.0009931674,0.00015979624,0.0011866553,0.0011036072,0.11656038,0.04584743,0.053651758,0.036007416,0.7338095],"study_design_scores_gemma":[0.00022503194,0.00049542426,0.0031412067,0.0001814858,0.00009195112,0.00090859586,0.00013812822,0.81966925,0.090575606,0.031138254,0.053324375,0.0001107918],"about_ca_topic_score_codex":0.0016485662,"about_ca_topic_score_gemma":0.0012999629,"teacher_disagreement_score":0.009864866,"about_ca_system_score_codex":0.0005082653,"about_ca_system_score_gemma":0.0012112262,"threshold_uncertainty_score":0.033001244},"labels":[],"label_agreement":null},{"id":"W2756297985","doi":"10.1007/978-3-319-67549-7_1","title":"Fragility-Oriented Testing with Model Execution and Reinforcement Learning","year":2017,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Fragility; Reinforcement learning; Computer science; Reinforcement; Stress testing (software); Artificial intelligence; Engineering; Operating system; Structural engineering","score_opus":0.02738793666073401,"score_gpt":0.2561419108886857,"score_spread":0.22875397422795168,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2756297985","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009562295,0.00021665813,0.98495966,0.00011240612,0.000027368975,0.00003741473,0.000020991167,0.0010989138,0.003964242],"genre_scores_gemma":[0.5392058,0.00029684807,0.45537412,0.00009236029,0.00004344377,0.00013378996,0.0001288115,0.00032357586,0.004401328],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99903107,0.00038079344,0.00004319626,0.00012959552,0.00033219575,0.000083230654],"domain_scores_gemma":[0.99690175,0.0022057584,0.00016561341,0.0004054073,0.00024813364,0.00007325776],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012529877,0.001110598,0.0006877715,0.0006989874,0.00026136095,0.0007055269,0.0018602762,0.0009026346,0.0035631845],"category_scores_gemma":[0.006046169,0.0004829761,0.0008584764,0.0005293546,0.00095431297,0.001515466,0.0013266533,0.0018504902,0.0004962036],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001570185,0.0001736282,0.00080520485,0.00016470956,0.00006869738,0.00012168271,0.00007831426,0.6971131,0.005045162,0.056960832,0.00200629,0.23730543],"study_design_scores_gemma":[0.0000101221385,0.000028129256,0.00007588744,0.000012175026,0.000008803446,0.00003637919,0.000004157922,0.9589361,0.0016039426,0.038701694,0.0005775215,0.000005125386],"about_ca_topic_score_codex":0.0012962735,"about_ca_topic_score_gemma":0.0016662913,"teacher_disagreement_score":0.0035631845,"about_ca_system_score_codex":0.00070331123,"about_ca_system_score_gemma":0.00070385204,"threshold_uncertainty_score":0.0119200945},"labels":[],"label_agreement":null},{"id":"W2760129307","doi":"10.1007/978-0-387-35516-0_20","title":"Erratum to: Testing of Communicating Systems","year":2017,"lang":"en","type":"erratum","venue":"IFIP advances in information and communication technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Library science; World Wide Web","score_opus":0.022167174185329402,"score_gpt":0.3033675095037852,"score_spread":0.2812003353184558,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2760129307","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00040150245,0.0016601684,0.001305117,0.04494191,0.94292873,0.00004488829,0.00040869752,0.00029280529,0.0080160815],"genre_scores_gemma":[0.024006296,0.022388736,0.013301455,0.14694925,0.31715906,0.00034924032,0.0043285973,0.0017631646,0.46975422],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949444,0.0007302728,0.00092820177,0.00038749707,0.0027073321,0.00030229834],"domain_scores_gemma":[0.9737091,0.0072810436,0.0011964558,0.0015887435,0.015388796,0.000835795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030024198,0.0018563281,0.001472184,0.0037559133,0.003653658,0.002843913,0.002668223,0.0076449914,0.034075413],"category_scores_gemma":[0.03612473,0.0008439316,0.001135502,0.0026123184,0.0022637695,0.0023096544,0.0015554122,0.0070287143,0.022290416],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031522835,0.000016750928,0.00006717269,0.00011565463,0.0000080636255,0.0003246805,0.000034859902,0.000067657515,0.00007357059,0.0015853867,0.98786366,0.009811092],"study_design_scores_gemma":[0.000037621743,0.00006549545,0.0007372414,0.00053262536,0.000047307916,0.000667306,0.00018052875,0.00042502154,0.0007058448,0.0028602872,0.9937017,0.000038910868],"about_ca_topic_score_codex":0.009637884,"about_ca_topic_score_gemma":0.013474917,"teacher_disagreement_score":0.034075413,"about_ca_system_score_codex":0.0029146716,"about_ca_system_score_gemma":0.0040564565,"threshold_uncertainty_score":0.113993585},"labels":[],"label_agreement":null},{"id":"W2760492644","doi":"10.1080/10556788.2019.1692344","title":"mts: a light framework for parallelizing tree search codes","year":2019,"lang":"en","type":"preprint","venue":"Optimization methods & software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Japan Society for the Promotion of Science","keywords":"Computer science; Backtracking; Enumeration; Satisfiability; Debugging; Search tree; Parallel computing; Vertex (graph theory); Theoretical computer science; Branch and bound; Search algorithm; Algorithm; Graph; Programming language; Discrete mathematics; Mathematics","score_opus":0.0714005721263264,"score_gpt":0.40162440456002696,"score_spread":0.33022383243370057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2760492644","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022599483,0.00008129245,0.98873353,0.00009955268,0.000043664062,0.0000861366,0.00011397139,0.006415589,0.002166359],"genre_scores_gemma":[0.06005082,0.00023524757,0.9307174,0.00017489334,0.00008642835,0.0006443101,0.00051031256,0.0036874935,0.0038932168],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9978325,0.00049881625,0.0002046193,0.00026905962,0.00091719226,0.00027787648],"domain_scores_gemma":[0.99752015,0.00077690044,0.00019603154,0.0008340459,0.00051401433,0.00015867036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016996129,0.0012750948,0.0010774148,0.0015758377,0.0012167982,0.0021070784,0.0031899333,0.0010963326,0.010450741],"category_scores_gemma":[0.0061235125,0.0010083383,0.0020786626,0.001968342,0.0024857302,0.0032692093,0.0030598887,0.0029376908,0.0035942567],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006002251,0.00017759447,0.0013371399,0.00063040666,0.00010021426,0.0004084515,0.00063659286,0.17673518,0.024201315,0.47968608,0.025030617,0.29045615],"study_design_scores_gemma":[0.0002135733,0.00014519015,0.00026642488,0.00011254135,0.00005733323,0.00021910017,0.00008906176,0.6332071,0.025048455,0.24940555,0.09116028,0.000075473945],"about_ca_topic_score_codex":0.00464805,"about_ca_topic_score_gemma":0.0052824565,"teacher_disagreement_score":0.010450741,"about_ca_system_score_codex":0.0017817816,"about_ca_system_score_gemma":0.0023900545,"threshold_uncertainty_score":0.034961224},"labels":[],"label_agreement":null},{"id":"W2767601065","doi":"10.1109/icsme.2017.21","title":"SimPact: Impact Analysis for Simulink Models","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; General Motors of Canada","keywords":"Computer science; Process (computing); Set (abstract data type); Reliability engineering; Model-based testing; Software engineering; Test case; Machine learning; Engineering; Programming language","score_opus":0.08757853160803547,"score_gpt":0.37799711569846617,"score_spread":0.2904185840904307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2767601065","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025809066,0.00019166523,0.9151371,0.00009616948,0.000047809266,0.0001727655,0.0017667294,0.05259193,0.004186731],"genre_scores_gemma":[0.46477222,0.00060658995,0.5211631,0.00009660034,0.000053832733,0.0006352214,0.0055223536,0.004181515,0.0029685516],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980568,0.00042314318,0.00019984988,0.00019732614,0.0010470412,0.00007581217],"domain_scores_gemma":[0.9883668,0.00780597,0.0013042894,0.0012666907,0.001149323,0.00010693193],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002128026,0.0019436659,0.00061615126,0.005235515,0.00046248914,0.0011709767,0.0013274299,0.0006263416,0.007958975],"category_scores_gemma":[0.017101405,0.0006311415,0.0013050926,0.0020376774,0.00050468533,0.001998125,0.0012793173,0.0010025965,0.001068806],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040461734,0.0002659943,0.021993242,0.0011739724,0.0003589771,0.0009965706,0.000567883,0.6136926,0.017270805,0.02869735,0.017304802,0.29727316],"study_design_scores_gemma":[0.000028827693,0.00007434742,0.0012974102,0.000055848086,0.000048228892,0.00018619084,0.000045903907,0.97212154,0.011146107,0.007963762,0.007006904,0.000024961071],"about_ca_topic_score_codex":0.0034652671,"about_ca_topic_score_gemma":0.0028452168,"teacher_disagreement_score":0.007958975,"about_ca_system_score_codex":0.00057239487,"about_ca_system_score_gemma":0.0010507136,"threshold_uncertainty_score":0.026625454},"labels":[],"label_agreement":null},{"id":"W2768693575","doi":"10.1109/issre.2017.34","title":"On FSM-Based Testing: An Empirical Study: Complete Round-Trip Versus Transition Trees","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Transition (genetics); Software testing; Empirical research; Programming language; Statistics; Mathematics; Chemistry; Software","score_opus":0.1997591464537318,"score_gpt":0.38378440311105216,"score_spread":0.18402525665732036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2768693575","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98871523,0.00067348965,0.008319038,0.00010272267,0.000016255226,0.00018637608,0.00037640528,0.00012799671,0.0014824936],"genre_scores_gemma":[0.9859795,0.00037283628,0.01203482,0.000051238156,0.000025645319,0.00012699308,0.00096194533,0.00006981792,0.00037729842],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9790684,0.011046745,0.0016077126,0.0019750004,0.0056726355,0.0006295281],"domain_scores_gemma":[0.6438204,0.31784248,0.0116326865,0.01360714,0.011414682,0.0016826062],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011409053,0.0008936606,0.000678885,0.0034428695,0.0005327879,0.0008982457,0.0016344081,0.0013743691,0.0010860277],"category_scores_gemma":[0.1388326,0.00032368975,0.00063724775,0.0028064784,0.001483844,0.0029221948,0.0012386917,0.0013157481,0.00029215895],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003504361,0.010165878,0.26573053,0.0031067398,0.0007178313,0.0014482086,0.005596656,0.18436524,0.01737204,0.0074582165,0.0061665135,0.49436787],"study_design_scores_gemma":[0.00046292332,0.015273087,0.27661434,0.0008314851,0.0006934145,0.0037629101,0.0049509807,0.65391237,0.025285551,0.007286999,0.010713207,0.00021274466],"about_ca_topic_score_codex":0.0025748045,"about_ca_topic_score_gemma":0.0029707584,"teacher_disagreement_score":0.011409053,"about_ca_system_score_codex":0.0011382186,"about_ca_system_score_gemma":0.00069045334,"threshold_uncertainty_score":0.060337603},"labels":[],"label_agreement":null},{"id":"W2770530563","doi":"10.1109/issrew.2017.51","title":"Finite State Machine Testing Complete Round-Trip Versus Transition Trees: On the Road of Finding the Most Effective Criterion","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Tree traversal; Finite-state machine; Computer science; Graph traversal; Cover (algebra); Tree (set theory); Random testing; Algorithm; Transition (genetics); Test case; Mathematics; Machine learning; Engineering","score_opus":0.08251322982207974,"score_gpt":0.3089630976320257,"score_spread":0.22644986780994592,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2770530563","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34084165,0.0056556175,0.64370275,0.002074294,0.0001456017,0.00052023615,0.0004060697,0.0021667052,0.0044871694],"genre_scores_gemma":[0.71082455,0.0008124883,0.28636816,0.00026475795,0.00007110549,0.00022925569,0.0005873408,0.00033026643,0.0005120587],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.974566,0.015632495,0.0013508851,0.0017820226,0.00585712,0.0008115471],"domain_scores_gemma":[0.8599531,0.114939705,0.0055352123,0.009222491,0.008260432,0.002089091],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012154193,0.001319685,0.0017720618,0.0033818253,0.0004839332,0.0020633442,0.00204251,0.0014654213,0.0016592297],"category_scores_gemma":[0.069644235,0.00039641227,0.0010673979,0.0018401018,0.0019608457,0.0060269004,0.0013814764,0.0018248935,0.00038857307],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00233687,0.0012302594,0.026678337,0.0023870866,0.00058118,0.0003620991,0.0006354419,0.21968919,0.03237829,0.053603504,0.0054128715,0.6547049],"study_design_scores_gemma":[0.00026958523,0.0038959247,0.012994884,0.00066727377,0.00030222235,0.0005962794,0.00074167945,0.8615815,0.039954063,0.07403201,0.004814136,0.00015041688],"about_ca_topic_score_codex":0.0013176943,"about_ca_topic_score_gemma":0.0021183456,"teacher_disagreement_score":0.012154193,"about_ca_system_score_codex":0.0013224455,"about_ca_system_score_gemma":0.0018505502,"threshold_uncertainty_score":0.064278305},"labels":[],"label_agreement":null},{"id":"W2787734695","doi":"10.22215/etd/2017-12062","title":"Towards Efficient Instrumentation for Reverse-Engineering Object Oriented Software through Static and Dynamic Analyses","year":2017,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Instrumentation (computer programming); Computer science; Static analysis; Overhead (engineering); Reverse engineering; Source code; C dynamic memory allocation; Software; Dynamic program analysis; Static program analysis; Context (archaeology); Program analysis; Software development; Programming language; Memory management","score_opus":0.03285469019548842,"score_gpt":0.35459371719772287,"score_spread":0.32173902700223445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2787734695","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010673074,0.00059543876,0.981834,0.00018544254,0.000043928627,0.000071780996,0.000052494215,0.004929631,0.0016141424],"genre_scores_gemma":[0.12663844,0.0010383507,0.86848277,0.00018310848,0.000044693716,0.00015803495,0.00036107042,0.0011252766,0.0019682546],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995063,0.001283774,0.0002494731,0.000613241,0.0023610867,0.00042948258],"domain_scores_gemma":[0.9912744,0.0032461926,0.0007130453,0.0029840756,0.0016466513,0.00013562439],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030054373,0.001090365,0.00091533683,0.0017675741,0.0004731026,0.0020999538,0.0017661209,0.0011955262,0.0015761924],"category_scores_gemma":[0.009751458,0.0008264478,0.00078260794,0.0015534733,0.001746689,0.003277863,0.0033577413,0.0028761635,0.0019454272],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023913046,0.00044419317,0.003911699,0.00070702704,0.00008557471,0.00029279687,0.0008262264,0.04046224,0.14932555,0.05807037,0.004276568,0.7413586],"study_design_scores_gemma":[0.00009528635,0.00035897727,0.002785379,0.00048781696,0.00016643798,0.0006953161,0.00037164247,0.5153267,0.30311072,0.12761936,0.048867118,0.00011528207],"about_ca_topic_score_codex":0.0009161278,"about_ca_topic_score_gemma":0.0010528925,"teacher_disagreement_score":0.0030054373,"about_ca_system_score_codex":0.0007069258,"about_ca_system_score_gemma":0.0017700891,"threshold_uncertainty_score":0.015894413},"labels":[],"label_agreement":null},{"id":"W2790194976","doi":"10.1016/bs.adcom.2018.01.001","title":"Model-Based Test Cases Reuse and Optimization","year":2018,"lang":"en","type":"book-chapter","venue":"Advances in computers","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Reuse; Reusability; Computer science; Redundancy (engineering); Test case; Reliability engineering; Model-based testing; Test (biology); Regression testing; Test Management Approach; Engineering; Machine learning; Software; Programming language; Software development; Software construction","score_opus":0.026656270361618307,"score_gpt":0.2734850027060155,"score_spread":0.2468287323443972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2790194976","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037632994,0.0009150488,0.9809598,0.00023216024,0.00007500076,0.00008077622,0.00007466179,0.0024641643,0.011435127],"genre_scores_gemma":[0.18370546,0.0017151792,0.794596,0.00023916333,0.00011305154,0.00026223302,0.00077174895,0.002421894,0.01617526],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9962179,0.0009156567,0.00021151887,0.00047698437,0.0019472924,0.00023064621],"domain_scores_gemma":[0.9943001,0.0028111564,0.00026540644,0.0018648558,0.00068996375,0.00006853322],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018841011,0.0018592674,0.0012276875,0.001984534,0.00038423348,0.0020122756,0.0032531745,0.0011694529,0.008474715],"category_scores_gemma":[0.011495995,0.0010859257,0.0022759119,0.0021009063,0.0010543515,0.0027956923,0.0019232404,0.0025738114,0.0022874898],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000111756526,0.00023509895,0.00074457715,0.0003828375,0.00014010565,0.00014394532,0.00013479768,0.18675542,0.009970179,0.111648515,0.010357359,0.6793754],"study_design_scores_gemma":[0.00003961114,0.000101463425,0.00054692605,0.00017003338,0.0001293901,0.00036747006,0.00003256273,0.8043722,0.016725225,0.15398136,0.023491027,0.000042616764],"about_ca_topic_score_codex":0.0023719256,"about_ca_topic_score_gemma":0.0025238923,"teacher_disagreement_score":0.008474715,"about_ca_system_score_codex":0.0013763994,"about_ca_system_score_gemma":0.0015006799,"threshold_uncertainty_score":0.02835071},"labels":[],"label_agreement":null},{"id":"W2790715424","doi":"10.1007/s10515-018-0232-y","title":"Black-box tree test case generation through diversity","year":2018,"lang":"en","type":"article","venue":"Automated Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Tree (set theory); Test case; Test (biology); Set (abstract data type); Black box; Algorithm; Random testing; Mathematics; Machine learning; Artificial intelligence","score_opus":0.024780093582989115,"score_gpt":0.24827839461829362,"score_spread":0.22349830103530452,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2790715424","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04572258,0.00012453728,0.946303,0.00016734688,0.000023575092,0.00021563514,0.0001622944,0.004174587,0.003106405],"genre_scores_gemma":[0.44724956,0.00007214117,0.54884785,0.00024998232,0.000024832012,0.0002184189,0.00063963496,0.0008253065,0.0018722571],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99586844,0.0015176295,0.0002336485,0.00067462487,0.0013112426,0.000394494],"domain_scores_gemma":[0.9864466,0.00873707,0.0006703557,0.0020712467,0.0016861325,0.00038844487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002764372,0.0011959443,0.0012053669,0.002771854,0.0006627486,0.0011865784,0.0026752008,0.0015444704,0.005953312],"category_scores_gemma":[0.014386263,0.0006706211,0.0012055436,0.0015220424,0.001242654,0.0023217394,0.0034114656,0.001456882,0.0013796752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011508702,0.0005433657,0.0076394635,0.00037006135,0.00019440374,0.0013768098,0.0007657132,0.1963238,0.04701985,0.029739797,0.0076074232,0.7072684],"study_design_scores_gemma":[0.00015953336,0.0002826997,0.0007225072,0.00006856362,0.00006943762,0.00056936755,0.000078595614,0.9325054,0.021640062,0.040292278,0.0035734598,0.000038116883],"about_ca_topic_score_codex":0.0012254553,"about_ca_topic_score_gemma":0.002009292,"teacher_disagreement_score":0.005953312,"about_ca_system_score_codex":0.0007754545,"about_ca_system_score_gemma":0.0011423653,"threshold_uncertainty_score":0.01991582},"labels":[],"label_agreement":null},{"id":"W2794522096","doi":"10.1145/3180155.3180203","title":"Fine-grained test minimization","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Test suite; Computer science; Test (biology); Code coverage; Test case; Test Management Approach; Minification; Test harness; Reliability engineering; Automatic test pattern generation; Test script; Test method; System under test; Algorithm; Model-based testing; Software; Programming language; Machine learning; Statistics; Mathematics; Software system; Engineering","score_opus":0.01795446925922184,"score_gpt":0.25962416323773274,"score_spread":0.2416696939785109,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2794522096","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1361599,0.0005167677,0.8557727,0.00030881306,0.000036805966,0.00041676805,0.00026030568,0.0036656356,0.0028622628],"genre_scores_gemma":[0.6096618,0.00017996224,0.38627368,0.00027687856,0.000028879513,0.0003680386,0.0010181204,0.0005975235,0.0015951497],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99451244,0.0017629648,0.00038636982,0.0009803247,0.0018914049,0.00046647937],"domain_scores_gemma":[0.9843321,0.008081048,0.0017376313,0.0041021616,0.0015616856,0.00018543165],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031603333,0.0011876306,0.0011642345,0.0026910002,0.0006001184,0.0011881584,0.0019874705,0.00067309814,0.0018669891],"category_scores_gemma":[0.01693986,0.00067657174,0.0016198709,0.0012557355,0.0012834332,0.0016978242,0.0017880738,0.00132428,0.0004030035],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004985153,0.0006331822,0.018683266,0.0006344515,0.0003131802,0.0005909496,0.00042591244,0.33129764,0.09289564,0.023728007,0.0036601082,0.52663916],"study_design_scores_gemma":[0.00009305107,0.00070216495,0.008606634,0.00011864831,0.00024008406,0.00080215686,0.00015135415,0.90201193,0.0478522,0.031831175,0.007528172,0.00006241912],"about_ca_topic_score_codex":0.0027233232,"about_ca_topic_score_gemma":0.0050615496,"teacher_disagreement_score":0.0031603333,"about_ca_system_score_codex":0.0010951437,"about_ca_system_score_gemma":0.0021596043,"threshold_uncertainty_score":0.01671362},"labels":[],"label_agreement":null},{"id":"W2795612311","doi":"10.1109/saner.2018.8330200","title":"Clustering support for inadequate test suite reduction","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Test suite; Cluster analysis; Regression testing; Computer science; Test (biology); Reduction (mathematics); Reliability engineering; Test case; Code coverage; Data mining; Test Management Approach; Suite; Automatic test pattern generation; Machine learning; Regression analysis; Engineering; Software; Programming language; Mathematics; Software system","score_opus":0.03564268704330998,"score_gpt":0.29757460533931246,"score_spread":0.2619319182960025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2795612311","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7456741,0.0024323016,0.20992038,0.0042248713,0.0002122664,0.00058803183,0.010398501,0.012599057,0.013950522],"genre_scores_gemma":[0.85637355,0.00034542405,0.12204022,0.00035420383,0.00010060613,0.00029650997,0.018907435,0.00047372768,0.0011083261],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9915701,0.002554784,0.0007010711,0.0014430189,0.003146413,0.00058462896],"domain_scores_gemma":[0.9264771,0.040409688,0.0064709177,0.012889838,0.01195359,0.0017988806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006398999,0.00091766665,0.0009312379,0.0052833785,0.0009099745,0.0019047263,0.0025882795,0.0016650677,0.002540948],"category_scores_gemma":[0.074558996,0.00036164367,0.00096748496,0.0044276337,0.00056663936,0.0018839999,0.00151959,0.0011301382,0.0010146725],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014493028,0.0010542148,0.16246116,0.0015039897,0.0005555926,0.0005699063,0.00079814123,0.18389492,0.022317735,0.014575747,0.071492895,0.5393263],"study_design_scores_gemma":[0.00024414006,0.0006782842,0.07044883,0.0002318271,0.00032099657,0.0008441501,0.0005740972,0.854348,0.01702818,0.022545611,0.0326162,0.00011961591],"about_ca_topic_score_codex":0.0048075872,"about_ca_topic_score_gemma":0.007907064,"teacher_disagreement_score":0.006398999,"about_ca_system_score_codex":0.0014381709,"about_ca_system_score_gemma":0.001964709,"threshold_uncertainty_score":0.03384161},"labels":[],"label_agreement":null},{"id":"W2796271057","doi":"10.1007/978-3-319-89363-1_13","title":"Iterative Generation of Diverse Models for Testing Specifications of DSL Tools","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Emberi Eroforrások Minisztériuma","keywords":"Computer science; Digital subscriber line; Test suite; Programming language; Iterative refinement; Model-based testing; Domain-specific language; Set (abstract data type); Generator (circuit theory); Suite; Software engineering; Context (archaeology); Graph; Code generation; Theoretical computer science; Test case; Algorithm; Machine learning","score_opus":0.19838911353524138,"score_gpt":0.30393732370412346,"score_spread":0.10554821016888208,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2796271057","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023741942,0.00014356124,0.97168523,0.00012619709,0.000015568903,0.00018853301,0.00010451226,0.0012376726,0.0027568266],"genre_scores_gemma":[0.19315723,0.00020970685,0.80322087,0.00008646524,0.000010351668,0.0003487734,0.0007310622,0.0007474581,0.0014881091],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.993694,0.0020939393,0.00031091648,0.00065670756,0.0030611237,0.00018323564],"domain_scores_gemma":[0.98868513,0.00695321,0.00047169882,0.0027539716,0.0009878636,0.00014801875],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038761217,0.0012335412,0.0007012787,0.0021749642,0.0006570222,0.001741815,0.0022794823,0.001393929,0.0019377862],"category_scores_gemma":[0.016112221,0.0010078321,0.0017682466,0.0012857536,0.0014123686,0.002833997,0.0033599841,0.0022819475,0.0007245747],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024614364,0.0003392041,0.004731105,0.00055645383,0.00016665389,0.00088980625,0.0017822111,0.33762887,0.06992618,0.14238474,0.003929819,0.4374189],"study_design_scores_gemma":[0.000045532273,0.00022780124,0.00059605687,0.00012620034,0.0000711357,0.00060981093,0.00024008346,0.85380363,0.055468556,0.0730821,0.01568218,0.000046930247],"about_ca_topic_score_codex":0.0006232736,"about_ca_topic_score_gemma":0.0014822715,"teacher_disagreement_score":0.0038761217,"about_ca_system_score_codex":0.0013945478,"about_ca_system_score_gemma":0.00085093855,"threshold_uncertainty_score":0.02049911},"labels":[],"label_agreement":null},{"id":"W2798496775","doi":"10.23919/date.2018.8342261","title":"Suspect set prediction in RTL bug hunting","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Debugging; Computer science; Algorithmic program debugging; Probabilistic logic; Set (abstract data type); Software bug; Representation (politics); Programming language; Graph; Theoretical computer science; Artificial intelligence; Software","score_opus":0.026260762018551712,"score_gpt":0.2771563603785227,"score_spread":0.250895598359971,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2798496775","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04892865,0.0004199468,0.94248164,0.000365858,0.000028369599,0.000097254735,0.00039802937,0.0066936812,0.0005865508],"genre_scores_gemma":[0.580875,0.00018454215,0.4162969,0.00014152698,0.000041811163,0.0001297567,0.0009627919,0.00037504153,0.0009926601],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973826,0.0006080435,0.00015842503,0.00057676027,0.0010246785,0.0002495265],"domain_scores_gemma":[0.9881039,0.0073983823,0.0014803158,0.0015442561,0.0011805956,0.00029248474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027335018,0.0013925007,0.0014470429,0.00447324,0.0007444176,0.0015778795,0.002873092,0.0017491034,0.0018598117],"category_scores_gemma":[0.016686855,0.0009703452,0.0012957606,0.0014929731,0.0012114778,0.0027397298,0.002086966,0.0016524256,0.00052478164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031437713,0.0003189228,0.022630727,0.00028715935,0.00017106895,0.000492719,0.00045929855,0.64302653,0.006948978,0.012152559,0.0052838246,0.3079138],"study_design_scores_gemma":[0.000012663492,0.000033694938,0.00069118507,0.000015429036,0.000015879503,0.000066282075,0.000019598998,0.9891045,0.001870253,0.007726708,0.00042994285,0.000013940301],"about_ca_topic_score_codex":0.009531314,"about_ca_topic_score_gemma":0.013321995,"teacher_disagreement_score":0.009531314,"about_ca_system_score_codex":0.0012198543,"about_ca_system_score_gemma":0.0021429402,"threshold_uncertainty_score":0.018951654},"labels":[],"label_agreement":null},{"id":"W2803276064","doi":"10.1145/3183440.3183470","title":"VISUFLOW","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science","score_opus":0.014934488957384216,"score_gpt":0.26813717810419996,"score_spread":0.25320268914681576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2803276064","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016190002,0.0016711412,0.22671585,0.0010593083,0.00050538767,0.0004930302,0.036722325,0.6301525,0.08649042],"genre_scores_gemma":[0.18789043,0.0023831776,0.29714254,0.0022278132,0.00023923238,0.001662665,0.16361612,0.19719477,0.14764325],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990231,0.000134082,0.000061839426,0.00022468713,0.00040435317,0.00015203354],"domain_scores_gemma":[0.99783105,0.0008265688,0.00014237787,0.00042822692,0.0006183003,0.00015345217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012224697,0.001524798,0.0005827037,0.0022948943,0.00079257647,0.0018919619,0.0022373302,0.0011300553,0.06407015],"category_scores_gemma":[0.007177833,0.0008626896,0.000852096,0.001104009,0.00046061774,0.0037807815,0.0024464128,0.0015123455,0.02689062],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000922056,0.0001469351,0.0036285545,0.0010490923,0.000052689447,0.0004150853,0.0008714284,0.0025268046,0.011977175,0.019489333,0.58766115,0.37125972],"study_design_scores_gemma":[0.00017510832,0.00016732154,0.0043305745,0.00042294982,0.0000435887,0.0007543356,0.000302614,0.021325098,0.022628915,0.0184033,0.9313044,0.00014177986],"about_ca_topic_score_codex":0.0051180925,"about_ca_topic_score_gemma":0.0066611115,"teacher_disagreement_score":0.06407015,"about_ca_system_score_codex":0.0010152197,"about_ca_system_score_gemma":0.0014603873,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2804918785","doi":"10.7939/r3pr7n043","title":"Diversity-Based Automated Test Case Generation","year":2015,"lang":"en","type":"article","venue":"University of Alberta Library","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alberta Innovates","keywords":"Test (biology); Diversity (politics); Computer science; Biology; Political science; Ecology","score_opus":0.031491472263569346,"score_gpt":0.2089218540776135,"score_spread":0.17743038181404414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2804918785","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054509033,0.00030036384,0.9375682,0.00014666596,0.0000335016,0.00032262682,0.00017791761,0.00394368,0.0029980652],"genre_scores_gemma":[0.4700134,0.0001877236,0.52633035,0.00010733246,0.000029155055,0.0005726579,0.0009626414,0.00038351788,0.0014132349],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99654704,0.0011561386,0.00018007123,0.00048575737,0.0013629283,0.0002680386],"domain_scores_gemma":[0.9908381,0.005696071,0.0006169406,0.0010982255,0.0015529466,0.0001976083],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019329314,0.0010980007,0.0010772881,0.0023582107,0.0005150619,0.0009502786,0.0017982809,0.0009678403,0.002864274],"category_scores_gemma":[0.010801928,0.00055080216,0.0012837126,0.001231658,0.0008756863,0.001062236,0.0016295307,0.000965078,0.00073186244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023335047,0.00018454682,0.0039975145,0.0002638645,0.000087516106,0.0005092681,0.00023194734,0.5110177,0.024757585,0.010720005,0.0034066832,0.44458994],"study_design_scores_gemma":[0.00007199124,0.0001265088,0.0005704467,0.000021367468,0.000026324476,0.00022533069,0.00003079557,0.98247635,0.0080882525,0.006370084,0.001973504,0.000019061723],"about_ca_topic_score_codex":0.002827638,"about_ca_topic_score_gemma":0.0027068907,"teacher_disagreement_score":0.002864274,"about_ca_system_score_codex":0.00089396,"about_ca_system_score_gemma":0.0013300198,"threshold_uncertainty_score":0.010222435},"labels":[],"label_agreement":null},{"id":"W2806092147","doi":"10.1007/s10270-018-0680-7","title":"Static slicing of Use Case Maps requirements models","year":2018,"lang":"en","type":"article","venue":"Software & Systems Modeling","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"King Fahd University of Petroleum and Minerals","keywords":"Slicing; Computer science; Computer graphics (images)","score_opus":0.12165666343186207,"score_gpt":0.3041852905049325,"score_spread":0.1825286270730704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806092147","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08987773,0.0003412574,0.8618004,0.00037446283,0.00007393976,0.00061660405,0.0027629572,0.01815254,0.026000105],"genre_scores_gemma":[0.62452406,0.00050393573,0.3610209,0.00012924489,0.000036388505,0.00035616726,0.004075407,0.0032289184,0.006125013],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9973066,0.00058980554,0.00015781341,0.0002905832,0.0013883535,0.00026685305],"domain_scores_gemma":[0.9921172,0.0037422169,0.0005501565,0.0019209292,0.0015533817,0.00011604466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002021674,0.0011306928,0.00081481336,0.0028757653,0.0006607846,0.0023290727,0.0013554347,0.00080104906,0.010423265],"category_scores_gemma":[0.010886666,0.0011296737,0.0015089752,0.0016105397,0.00086292304,0.0029709535,0.0018931563,0.001191693,0.0011083806],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006253051,0.00032279635,0.010779254,0.0011689096,0.0002079715,0.0024579333,0.003985708,0.3253021,0.031212823,0.19022085,0.01599971,0.41771662],"study_design_scores_gemma":[0.000045939658,0.00008870797,0.0024587137,0.0003481338,0.00016217433,0.00037733637,0.00048141595,0.8694669,0.029738888,0.06459227,0.032170277,0.0000692108],"about_ca_topic_score_codex":0.018542152,"about_ca_topic_score_gemma":0.023327988,"teacher_disagreement_score":0.018542152,"about_ca_system_score_codex":0.0014708044,"about_ca_system_score_gemma":0.002622909,"threshold_uncertainty_score":0.036868453},"labels":[],"label_agreement":null},{"id":"W2807011628","doi":"10.1007/978-3-319-92997-2_13","title":"Life Sciences-Inspired Test Case Similarity Measures for Search-Based, FSM-Based Software Testing","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Test suite; Computer science; Suite; Intuition; Diversity (politics); Machine learning; Fault detection and isolation; Random testing; Software; Test case; Test (biology); Artificial intelligence; Data mining; Software engineering; Programming language; Psychology","score_opus":0.08310386132228398,"score_gpt":0.3036598676674446,"score_spread":0.22055600634516065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807011628","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09374949,0.0020933598,0.895611,0.00027784356,0.00011092103,0.0001986056,0.00036030132,0.0011752049,0.0064232918],"genre_scores_gemma":[0.64313656,0.0004012858,0.3536174,0.00010531199,0.000089222216,0.00020675598,0.00090485153,0.00015369036,0.0013849993],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968894,0.00074597937,0.00024404639,0.0003438292,0.0016576457,0.00011896468],"domain_scores_gemma":[0.9927799,0.0039876215,0.0008856324,0.00075816055,0.0013369285,0.00025170355],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001758412,0.00067086343,0.0010393847,0.0048827752,0.0004436348,0.001416968,0.001410924,0.0009812785,0.0016391425],"category_scores_gemma":[0.015357227,0.00020889204,0.0008695459,0.0030142013,0.00078803534,0.002056699,0.0012838733,0.0009997555,0.0003335356],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004276684,0.0005823003,0.011632025,0.000678305,0.00026200962,0.0002183824,0.00033063933,0.1582868,0.022037398,0.08115918,0.0045502693,0.71983504],"study_design_scores_gemma":[0.000021221,0.00029620534,0.0052060974,0.00006544762,0.000056715107,0.00035788913,0.00008310204,0.9249686,0.0075053424,0.058190446,0.003220571,0.000028348077],"about_ca_topic_score_codex":0.0010058518,"about_ca_topic_score_gemma":0.0012222193,"teacher_disagreement_score":0.0048827752,"about_ca_system_score_codex":0.0016273771,"about_ca_system_score_gemma":0.000900273,"threshold_uncertainty_score":0.011807442},"labels":[],"label_agreement":null},{"id":"W2808023677","doi":"10.1145/3183519.3183551","title":"Evaluating specification-level MC/DC criterion in model-based testing of safety critical systems","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Microsoft (Canada)","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Code coverage; Certification; Reliability engineering; Model-based testing; Process (computing); White-box testing; Software; Software engineering; Software system; Test case; Programming language; Software construction; Engineering","score_opus":0.2588067781531558,"score_gpt":0.40762558216868033,"score_spread":0.14881880401552455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2808023677","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44374055,0.0006471671,0.54205537,0.00041564603,0.000038968603,0.00037827663,0.0002986045,0.0013809602,0.011044403],"genre_scores_gemma":[0.9364982,0.00007531751,0.062226184,0.00009485145,0.000014371623,0.00015201907,0.0003784034,0.00013343671,0.00042716853],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9824441,0.008550942,0.00067495246,0.0010853604,0.0060656634,0.001178913],"domain_scores_gemma":[0.90497285,0.07830208,0.0029730215,0.004674836,0.00782282,0.0012544306],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011613853,0.0009098389,0.0008848352,0.004126378,0.0006104812,0.002293182,0.001973746,0.0015946621,0.0023015924],"category_scores_gemma":[0.08715429,0.00045689018,0.0012384263,0.0012733635,0.0016933447,0.0023031258,0.001708269,0.000934341,0.00027942957],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007526081,0.0004602813,0.030454407,0.00040228982,0.00016224709,0.0003166985,0.00022045577,0.8613667,0.013621243,0.038532488,0.0010517356,0.052658867],"study_design_scores_gemma":[0.000035009896,0.0002118604,0.0013518323,0.00003128546,0.000026644897,0.0000513349,0.00003948235,0.9893798,0.0037287078,0.0048598573,0.00027344076,0.000010702934],"about_ca_topic_score_codex":0.016220538,"about_ca_topic_score_gemma":0.010174655,"teacher_disagreement_score":0.016220538,"about_ca_system_score_codex":0.0034742267,"about_ca_system_score_gemma":0.0038176882,"threshold_uncertainty_score":0.06142068},"labels":[],"label_agreement":null},{"id":"W2809981234","doi":"10.1002/stvr.1665","title":"P<scp>esto</scp>: Automated migration of DOM‐based Web tests towards the visual approach","year":2018,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Automation; Web application; Test (biology); Visual Basic; Visualization; Software engineering; Artificial intelligence; Programming language; Software; World Wide Web; Engineering","score_opus":0.033805008270209574,"score_gpt":0.28879608613740154,"score_spread":0.254991077867192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2809981234","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06062366,0.000089150715,0.7077764,0.00027024795,0.000114356335,0.00033563966,0.0015813595,0.2209654,0.008243709],"genre_scores_gemma":[0.5276784,0.00012227561,0.43895453,0.00042318678,0.000050951156,0.0005160249,0.0064631877,0.016246004,0.009545411],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99817777,0.0004663448,0.0002186325,0.00029772153,0.0006835867,0.00015600333],"domain_scores_gemma":[0.9923086,0.0023844915,0.00062668824,0.0029460308,0.0014284017,0.00030577672],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017495062,0.0012307152,0.00055310037,0.001747122,0.000417167,0.0014616555,0.0016849074,0.00091167685,0.0050389194],"category_scores_gemma":[0.0080501335,0.00062432955,0.00069003657,0.00059765653,0.0007978069,0.0014213177,0.001554424,0.0010861006,0.002631515],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001383808,0.00071207137,0.021548178,0.00068080006,0.00017873036,0.0037191007,0.0007477165,0.05693682,0.13130903,0.011082459,0.10386644,0.6678348],"study_design_scores_gemma":[0.00025541216,0.0004348807,0.0111313425,0.0001604301,0.000059802744,0.0018636078,0.00013430798,0.69050306,0.22929154,0.006944518,0.059066087,0.00015510188],"about_ca_topic_score_codex":0.002597894,"about_ca_topic_score_gemma":0.0020345333,"teacher_disagreement_score":0.0050389194,"about_ca_system_score_codex":0.0004294301,"about_ca_system_score_gemma":0.0010916257,"threshold_uncertainty_score":0.01685685},"labels":[],"label_agreement":null},{"id":"W2810463508","doi":"10.1109/mtv.2017.19","title":"Dynamic Exerciser Template Weighting in x86 Processor Verification","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Advanced Micro Devices (Canada)","funders":"","keywords":"Computer science; x86; Template; Weighting; Functional verification; Embedded system; Computer engineering; Theoretical computer science; Formal verification; Software; Operating system; Programming language","score_opus":0.021446376472997554,"score_gpt":0.3025776588844818,"score_spread":0.2811312824114842,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2810463508","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13427898,0.0002867139,0.8578972,0.00009226302,0.00003630182,0.00018502693,0.000040943418,0.00467926,0.0025033392],"genre_scores_gemma":[0.701997,0.00009346074,0.29517218,0.00009296684,0.000021465632,0.0001250767,0.00009347001,0.0005759828,0.0018283244],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965655,0.0011089434,0.0002647028,0.0006805211,0.00110866,0.0002716862],"domain_scores_gemma":[0.99379045,0.00279935,0.0007235141,0.0019055497,0.0005689796,0.0002121285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027544952,0.00064590253,0.00050109596,0.0010400341,0.00038441352,0.00085652154,0.0016257293,0.00063854817,0.0021149646],"category_scores_gemma":[0.01276177,0.0004482579,0.00036816997,0.00069795165,0.00089322816,0.0020821146,0.0016084547,0.0007117871,0.00063389284],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001042455,0.0004130851,0.008704312,0.0002300202,0.00005794016,0.00027106985,0.0007514203,0.09686167,0.12454917,0.01801828,0.00134932,0.74775136],"study_design_scores_gemma":[0.00016881381,0.0011895847,0.00431171,0.00007681456,0.00008020319,0.0006176672,0.00012044611,0.7812505,0.17843583,0.021504665,0.012155678,0.00008813],"about_ca_topic_score_codex":0.0006746809,"about_ca_topic_score_gemma":0.0011548662,"teacher_disagreement_score":0.0027544952,"about_ca_system_score_codex":0.00054524495,"about_ca_system_score_gemma":0.00072505063,"threshold_uncertainty_score":0.014567316},"labels":[],"label_agreement":null},{"id":"W2810772136","doi":"10.5539/mas.v12n7p99","title":"Using Artificial Bee Colony Algorithm for Test Data Generation and Path Testing Coverage","year":2018,"lang":"en","type":"article","venue":"Modern Applied Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Traverse; Keyword-driven testing; White-box testing; Automation; Software; Path (computing); Code coverage; Heuristic; Artificial bee colony algorithm; Test strategy; Data mining; Software development; Artificial intelligence; Software construction; Programming language; Engineering","score_opus":0.18411092876477508,"score_gpt":0.33815538343728446,"score_spread":0.15404445467250938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2810772136","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08655931,0.00035144985,0.90694946,0.00027868204,0.00004925447,0.00016048097,0.000064411586,0.0010484202,0.0045384564],"genre_scores_gemma":[0.68396044,0.00022713843,0.31250292,0.00014592892,0.000018588804,0.00033587424,0.00022213846,0.00007787137,0.0025090375],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99954826,0.00012088743,0.000032135616,0.00008433044,0.0001655654,0.000048820493],"domain_scores_gemma":[0.99923754,0.0004067075,0.00007320274,0.000053739022,0.00020554115,0.000023169714],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00052040117,0.000602946,0.00059903605,0.00079754996,0.00031599705,0.0005899691,0.0010094288,0.00078391255,0.00095258915],"category_scores_gemma":[0.0024285065,0.00023649225,0.0004663508,0.0006086749,0.00034557708,0.00056449685,0.0004622042,0.0005259033,0.00012595393],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007699332,0.00012128793,0.0036251405,0.00013110426,0.00008668628,0.00022030374,0.0001626298,0.7612711,0.014982889,0.005327397,0.0017252158,0.21226922],"study_design_scores_gemma":[0.00001621722,0.000048676193,0.00040470835,0.000008366839,0.000013918605,0.000049314407,0.000016406537,0.9949576,0.0024641033,0.0012317126,0.00078227563,0.000006660931],"about_ca_topic_score_codex":0.004684792,"about_ca_topic_score_gemma":0.0029878404,"teacher_disagreement_score":0.004684792,"about_ca_system_score_codex":0.00043449606,"about_ca_system_score_gemma":0.0007299776,"threshold_uncertainty_score":0.0093150735},"labels":[],"label_agreement":null},{"id":"W2867457158","doi":"10.1145/3213846.3213847","title":"Safe and sound program analysis with Flix","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Sound (geography); Acoustics; Physics","score_opus":0.016017412333590296,"score_gpt":0.28403457255650233,"score_spread":0.26801716022291205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2867457158","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031943892,0.00024924948,0.9772246,0.00065091345,0.000061002273,0.000045282188,0.00016349595,0.01188272,0.0065285247],"genre_scores_gemma":[0.16355084,0.00071420125,0.81503665,0.0010311092,0.00018123533,0.00024330593,0.0009413593,0.004565031,0.013736335],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99552166,0.00093964185,0.00027246855,0.0008053403,0.0019059732,0.0005549579],"domain_scores_gemma":[0.9938252,0.0032505118,0.00042833452,0.00141404,0.00095716,0.00012479893],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004629272,0.0014083956,0.0009387978,0.002484306,0.001211453,0.0039143343,0.001830467,0.0015234374,0.013012343],"category_scores_gemma":[0.016163312,0.001068851,0.0016091557,0.0010636577,0.004079339,0.0061850157,0.004008882,0.004532567,0.0049576443],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005279961,0.00017350093,0.0026143899,0.0007291052,0.00017540017,0.00040153656,0.00108686,0.029115248,0.018910047,0.43328336,0.026337385,0.48664513],"study_design_scores_gemma":[0.00015851535,0.000113376285,0.00060929137,0.000548589,0.0001350054,0.00033044213,0.0002647762,0.13891979,0.055435494,0.7356362,0.067741156,0.00010742238],"about_ca_topic_score_codex":0.0029587522,"about_ca_topic_score_gemma":0.0033564474,"teacher_disagreement_score":0.013012343,"about_ca_system_score_codex":0.0014220285,"about_ca_system_score_gemma":0.0033460238,"threshold_uncertainty_score":0.043530643},"labels":[],"label_agreement":null},{"id":"W2884350287","doi":"10.1109/icstw.2018.00037","title":"Test Automation - Automation of What?","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Automation; Software deployment; Computer science; Test (biology); Software engineering; Software; Engineering; Programming language","score_opus":0.017588103902229337,"score_gpt":0.2754708383392292,"score_spread":0.2578827344369999,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2884350287","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030973997,0.009063547,0.8183428,0.052522443,0.0013671027,0.00032291835,0.00028356988,0.004452548,0.08267108],"genre_scores_gemma":[0.7062572,0.0052054063,0.26406637,0.0081150085,0.0013409749,0.00034278492,0.0003450148,0.001088806,0.013238344],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98613966,0.0069713024,0.00074868614,0.0015818896,0.0037298147,0.00082861097],"domain_scores_gemma":[0.9640515,0.021497093,0.001894686,0.008032015,0.003534324,0.000990404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009956521,0.0012735637,0.0010833916,0.001574391,0.0013916548,0.006121934,0.0021185242,0.0028166212,0.00412388],"category_scores_gemma":[0.03855169,0.00061148364,0.0010595743,0.0015536745,0.011201398,0.014300379,0.0022147272,0.005457983,0.0022404224],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022740835,0.00024431734,0.0061563756,0.0009611631,0.000075990574,0.00072231604,0.0047883918,0.0059948126,0.00832422,0.64433974,0.019603716,0.30856153],"study_design_scores_gemma":[0.000046669855,0.0002252067,0.003111929,0.0006604187,0.00004858213,0.0015245232,0.0021050482,0.018229038,0.009903985,0.87284464,0.091199465,0.00010056021],"about_ca_topic_score_codex":0.0019580743,"about_ca_topic_score_gemma":0.001117164,"teacher_disagreement_score":0.009956521,"about_ca_system_score_codex":0.0011001702,"about_ca_system_score_gemma":0.0019278692,"threshold_uncertainty_score":0.052655756},"labels":[],"label_agreement":null},{"id":"W2886792510","doi":"10.1109/qrs.2018.00056","title":"Avoiding the Familiar to Speed Up Test Case Reduction","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Test suite; Reduction (mathematics); Computer science; Debugging; Fuzz testing; Compiler; Test case; Test (biology); Process (computing); Programming language; Algorithm; Suite; Parallel computing; Computer engineering; Software; Machine learning; Mathematics","score_opus":0.0372376855372135,"score_gpt":0.293992681587254,"score_spread":0.2567549960500405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2886792510","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.120295025,0.0010705303,0.8452855,0.0014195632,0.00016023772,0.0006483327,0.0002883129,0.01959416,0.011238357],"genre_scores_gemma":[0.3677427,0.00035077124,0.62200373,0.00077145285,0.000074687,0.00038441084,0.00069504726,0.0029751677,0.0050019617],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9880747,0.005032154,0.0006661103,0.0013442215,0.0041539967,0.00072877906],"domain_scores_gemma":[0.9098076,0.06326119,0.0035448403,0.01756343,0.005068344,0.0007546742],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048316587,0.002133982,0.0011728163,0.0027585747,0.0009364197,0.0018542247,0.0032932144,0.0012209773,0.007959547],"category_scores_gemma":[0.054450866,0.0011610548,0.0014485904,0.0020486924,0.0018407964,0.004623113,0.003311181,0.0030414273,0.0030881208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088194903,0.00067157403,0.013965606,0.0010031138,0.00018041485,0.00060235575,0.0011488649,0.057704598,0.056119848,0.02440474,0.01419204,0.82912487],"study_design_scores_gemma":[0.00036280634,0.0010939761,0.010758202,0.00046323828,0.00031177557,0.0021648202,0.0006217836,0.7626828,0.10516973,0.07267221,0.04350209,0.00019653971],"about_ca_topic_score_codex":0.0036103912,"about_ca_topic_score_gemma":0.007762468,"teacher_disagreement_score":0.007959547,"about_ca_system_score_codex":0.0012613636,"about_ca_system_score_gemma":0.0028690652,"threshold_uncertainty_score":0.026627302},"labels":[],"label_agreement":null},{"id":"W2895807070","doi":"10.1145/3273934.3273944","title":"An Improvement to Test Case Failure Prediction in the Context of Test Case Prioritization","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); Toronto Metropolitan University","funders":"","keywords":"Computer science; Test (biology); Test suite; Ranking (information retrieval); Context (archaeology); Test plan; Machine learning; Test case; A priori and a posteriori; Scale (ratio); Predictive modelling; Data mining; Logistic regression; Reliability (semiconductor); Plan (archaeology); Artificial intelligence; Reliability engineering; Regression analysis; Statistics; Engineering","score_opus":0.014813174640073286,"score_gpt":0.27506018310741376,"score_spread":0.26024700846734045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2895807070","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13272966,0.0026577748,0.8456762,0.0048585674,0.0003728139,0.0005222702,0.0024854557,0.007111472,0.0035858192],"genre_scores_gemma":[0.67555416,0.0007320447,0.31814355,0.0006315005,0.00043686526,0.0002471142,0.0028530217,0.00031790248,0.0010837818],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9892609,0.005151425,0.0007099504,0.00272593,0.0018094069,0.0003423998],"domain_scores_gemma":[0.9109246,0.065522455,0.005326409,0.008882963,0.008515402,0.0008282391],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012736743,0.0026843972,0.0017005451,0.004071419,0.0006519725,0.0025523943,0.0036825305,0.0017387301,0.002832822],"category_scores_gemma":[0.08917016,0.0007750346,0.0019498448,0.0038968467,0.00071621465,0.004750398,0.0017090401,0.0041700406,0.0013664047],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009592696,0.0013929997,0.112067744,0.0008687533,0.00090088195,0.00041196987,0.0005221493,0.2958758,0.0038798107,0.0045429454,0.010901889,0.56767577],"study_design_scores_gemma":[0.00005408046,0.00026122972,0.008074513,0.0000748029,0.00013373609,0.00019738468,0.00007402172,0.9838023,0.0011629043,0.0039095664,0.002213118,0.000042330677],"about_ca_topic_score_codex":0.013779714,"about_ca_topic_score_gemma":0.011104609,"teacher_disagreement_score":0.013779714,"about_ca_system_score_codex":0.0012977726,"about_ca_system_score_gemma":0.0022751605,"threshold_uncertainty_score":0.06735909},"labels":[],"label_agreement":null},{"id":"W2896097868","doi":"10.1109/modre.2018.00007","title":"Modelling and Testing Requirements via Executable Abstract State Machines","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Eiffel; Computer science; Executable; Programming language; Formal specification; Notation; Software engineering; Object-oriented programming; Mathematics","score_opus":0.07012389338083574,"score_gpt":0.29371774342755136,"score_spread":0.22359385004671561,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2896097868","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0044558058,0.000021147067,0.9922144,0.00004452849,0.000009490604,0.000073789684,0.00008381617,0.0016233972,0.0014735737],"genre_scores_gemma":[0.118330024,0.00013961583,0.87751067,0.000067943656,0.000020681447,0.00043584325,0.0005571866,0.0004947173,0.002443336],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.992488,0.0026228956,0.00060155796,0.0007457518,0.0032318921,0.00030999293],"domain_scores_gemma":[0.9898587,0.005741529,0.00086888304,0.0023768602,0.0010245426,0.00012945486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037354557,0.001124171,0.00056159677,0.0012244823,0.00054590695,0.0023669368,0.0018969082,0.0015477269,0.0037362117],"category_scores_gemma":[0.015095581,0.0009234028,0.0015110917,0.00074215484,0.0020612902,0.00405092,0.0013853353,0.0017528347,0.0014446548],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018344758,0.00027523912,0.0020203663,0.00054144196,0.00007956461,0.0011461835,0.0016128254,0.22701438,0.040414147,0.60148984,0.003150445,0.12207213],"study_design_scores_gemma":[0.000074773845,0.00017741953,0.0004037559,0.00015876447,0.00006778422,0.0005309073,0.0001283172,0.6855069,0.06084634,0.21759568,0.03444893,0.000060476203],"about_ca_topic_score_codex":0.0018648658,"about_ca_topic_score_gemma":0.0019546351,"teacher_disagreement_score":0.0037362117,"about_ca_system_score_codex":0.00084995566,"about_ca_system_score_gemma":0.0015190098,"threshold_uncertainty_score":0.019755185},"labels":[],"label_agreement":null},{"id":"W2898893292","doi":"10.1145/3236024.3236063","title":"Visual web test repair","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":76,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Focus (optics); Regression testing; Workflow; Pipeline (software); Test (biology); Test case; Artificial intelligence; Machine learning; Database; Software; Programming language; Regression analysis; Software development","score_opus":0.01627182758737435,"score_gpt":0.28774382715051855,"score_spread":0.2714719995631442,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2898893292","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06333788,0.0009371378,0.62314636,0.0006594399,0.0003030234,0.0011188951,0.0029325509,0.29040202,0.017162643],"genre_scores_gemma":[0.5418953,0.0006184728,0.4114246,0.00070171035,0.00015761727,0.0007593436,0.008689772,0.016436229,0.019317023],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9966504,0.0009045973,0.00028088724,0.0006542132,0.0012865346,0.00022334371],"domain_scores_gemma":[0.9727692,0.009643225,0.0030588044,0.00978168,0.004082157,0.0006650574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030788379,0.0018863839,0.0009166075,0.0044065136,0.000550651,0.00232436,0.0029883464,0.0014900813,0.012383406],"category_scores_gemma":[0.027726255,0.00078406796,0.0010036717,0.0016299821,0.00079601124,0.0023106334,0.0026607544,0.0013535692,0.0044963686],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007905757,0.00058434973,0.011382765,0.0011153956,0.00017899048,0.0013394662,0.0013458596,0.023069145,0.03973277,0.00551497,0.07089002,0.8440557],"study_design_scores_gemma":[0.0004314879,0.0015511425,0.03259565,0.00086277333,0.00025537642,0.005956083,0.001735735,0.52570856,0.19714351,0.025145248,0.2082517,0.000362862],"about_ca_topic_score_codex":0.0027026178,"about_ca_topic_score_gemma":0.0026339865,"teacher_disagreement_score":0.012383406,"about_ca_system_score_codex":0.0007946308,"about_ca_system_score_gemma":0.00094808365,"threshold_uncertainty_score":0.0414266},"labels":[],"label_agreement":null},{"id":"W2900453375","doi":"10.1109/icsme.2018.00016","title":"Test Re-Prioritization in Continuous Testing Environments","year":2018,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Prioritization; Scalability; Reliability engineering; Set (abstract data type); Regression testing; Test case; Test strategy; Software deployment; Test (biology); Risk-based testing; Data mining; Software; Machine learning; Software system; Engineering","score_opus":0.026343412774647586,"score_gpt":0.25770566259295047,"score_spread":0.2313622498183029,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2900453375","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17404076,0.0034346795,0.7774249,0.0011970374,0.0003204693,0.000740672,0.00094292534,0.035380054,0.0065185474],"genre_scores_gemma":[0.65925133,0.0004612762,0.33261758,0.0007809172,0.00015560034,0.00029930146,0.0018462719,0.0014271563,0.0031606462],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9927429,0.001758947,0.0004767572,0.0017604843,0.0025814536,0.0006794207],"domain_scores_gemma":[0.96717995,0.017622465,0.002733108,0.006993754,0.0038060732,0.0016646034],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073154476,0.002083504,0.00112209,0.0029528544,0.000780131,0.0022591173,0.0038203984,0.0009294221,0.0020419445],"category_scores_gemma":[0.023892019,0.00089245534,0.00086353254,0.0015415965,0.0011911182,0.0031006327,0.0018775433,0.0021416864,0.0009808368],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014815032,0.0007261284,0.03885618,0.00074013125,0.0002355588,0.0007957665,0.0009481819,0.08984255,0.05699372,0.01050756,0.016904358,0.7819683],"study_design_scores_gemma":[0.0005002624,0.00096577255,0.026640283,0.00022500081,0.00027414982,0.0017905907,0.0006582166,0.8385207,0.06561975,0.03334014,0.031261407,0.00020370346],"about_ca_topic_score_codex":0.009604407,"about_ca_topic_score_gemma":0.013086991,"teacher_disagreement_score":0.009604407,"about_ca_system_score_codex":0.001320423,"about_ca_system_score_gemma":0.0032356756,"threshold_uncertainty_score":0.038688242},"labels":[],"label_agreement":null},{"id":"W2901173357","doi":"10.1007/978-3-030-03592-1_2","title":"Executable Counterexamples in Software Model Checking","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Executable; Computer science; Programming language; Counterexample; Model checking; Software; Software engineering; Operating system; Discrete mathematics","score_opus":0.035696260353724835,"score_gpt":0.2674783825539967,"score_spread":0.23178212220027186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2901173357","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04899308,0.0024879763,0.9246115,0.0012358073,0.00038861192,0.000089930436,0.00008568432,0.001851862,0.020255642],"genre_scores_gemma":[0.71858305,0.0012887986,0.26869634,0.00040784385,0.00022731935,0.00021063481,0.00028134865,0.0009082747,0.009396323],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99452776,0.0023986923,0.0002497892,0.0004925948,0.0020690793,0.00026212534],"domain_scores_gemma":[0.98237026,0.015096962,0.00040618895,0.0014136224,0.000598595,0.0001143785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026306843,0.0016595032,0.0011776468,0.0021895792,0.00091352063,0.002836218,0.0021407786,0.0021502941,0.0035453595],"category_scores_gemma":[0.02549235,0.0014419045,0.0012812775,0.001777739,0.004162752,0.005894029,0.0029610898,0.0051880935,0.000580215],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030865456,0.00012735621,0.00081449345,0.0004893879,0.000068552334,0.0004495071,0.0005048503,0.06617974,0.0025955888,0.82262504,0.0042628837,0.10157391],"study_design_scores_gemma":[0.000033881035,0.000036486577,0.00013159674,0.00016146789,0.000049627368,0.00020895098,0.000048673344,0.2111758,0.0042530517,0.77926654,0.0046103345,0.000023517818],"about_ca_topic_score_codex":0.001159787,"about_ca_topic_score_gemma":0.001623708,"teacher_disagreement_score":0.0035453595,"about_ca_system_score_codex":0.0014320564,"about_ca_system_score_gemma":0.0010262257,"threshold_uncertainty_score":0.013912559},"labels":[],"label_agreement":null},{"id":"W2903495827","doi":"10.1016/j.jss.2020.110542","title":"On testing machine learning programs","year":2020,"lang":"en","type":"preprint","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Transformative learning; Domain (mathematical analysis); Witness; Software engineering; Field (mathematics); Reliability (semiconductor); Software testing; Artificial intelligence; Software; Data science; Machine learning; Psychology","score_opus":0.06147200893269984,"score_gpt":0.2724679788593244,"score_spread":0.21099596992662456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903495827","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.066932246,0.0040217144,0.8977884,0.0050438256,0.00072524126,0.0001866457,0.00039385507,0.0023297127,0.022578403],"genre_scores_gemma":[0.69198817,0.0021659322,0.28659648,0.001613507,0.0014100067,0.00031297372,0.001666881,0.0012342382,0.0130117815],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9843576,0.0077919415,0.0009528748,0.0018808168,0.0040249815,0.0009917675],"domain_scores_gemma":[0.82953256,0.14874014,0.0017534047,0.014405817,0.0044298596,0.0011382508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074770977,0.003147743,0.002845463,0.005191264,0.001364154,0.0021950149,0.0050300932,0.0037945043,0.009724172],"category_scores_gemma":[0.08197311,0.001592562,0.0020863365,0.00488133,0.0065403897,0.012563232,0.0061706994,0.005180425,0.00088659103],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014544093,0.0006768494,0.009906262,0.0007581528,0.00036189787,0.00073211675,0.00053578056,0.20251967,0.0049930457,0.21585612,0.02385618,0.5383495],"study_design_scores_gemma":[0.00010962837,0.0001727905,0.0020486247,0.00017727603,0.00015202576,0.00025316275,0.000113251575,0.4859421,0.0034386658,0.50362086,0.0039377487,0.0000338435],"about_ca_topic_score_codex":0.005622361,"about_ca_topic_score_gemma":0.006447045,"teacher_disagreement_score":0.009724172,"about_ca_system_score_codex":0.0025664405,"about_ca_system_score_gemma":0.0017080463,"threshold_uncertainty_score":0.03954315},"labels":[],"label_agreement":null},{"id":"W2906640661","doi":"10.1016/bs.adcom.2018.10.005","title":"Three Open Problems in the Context of E2E Web Testing and a Vision: NEONATE","year":2018,"lang":"en","type":"book-chapter","venue":"Advances in computers","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Automation; Computer science; Web testing; Cohesion (chemistry); Web application; Software engineering; Context (archaeology); Web standards; Quality (philosophy); Web modeling; Data science; Web service; Web development; World Wide Web; Web application security; Engineering","score_opus":0.040664011614605776,"score_gpt":0.30901489416049815,"score_spread":0.2683508825458924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2906640661","genre_codex":"commentary","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007398579,0.12438764,0.3006107,0.31952736,0.008718375,0.00011164124,0.00021559074,0.002259513,0.23677066],"genre_scores_gemma":[0.20128085,0.116083935,0.39575157,0.053388633,0.009718749,0.0004524806,0.0005409083,0.0023309847,0.22045189],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9948895,0.0021111306,0.00021999073,0.00049011305,0.0018670469,0.00042224547],"domain_scores_gemma":[0.982746,0.010972564,0.00037288215,0.0016096231,0.0029147123,0.0013841508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009982915,0.0008121127,0.0010097625,0.0022710597,0.0027835353,0.011544593,0.0029984687,0.007291545,0.01164066],"category_scores_gemma":[0.021106506,0.0006082458,0.00075657124,0.0024734994,0.008806883,0.025822148,0.007994458,0.011089017,0.0036642603],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003058461,0.00006151616,0.0002997534,0.00020862652,0.000005460109,0.00020556334,0.0008675526,0.00051168347,0.00026897562,0.76203376,0.0720419,0.16346464],"study_design_scores_gemma":[0.00001132879,0.000044263026,0.00021169132,0.0008242891,0.0000083706145,0.0008308999,0.0011635254,0.0039776387,0.00080412783,0.6357898,0.3562869,0.000047190148],"about_ca_topic_score_codex":0.0032935396,"about_ca_topic_score_gemma":0.0044828295,"teacher_disagreement_score":0.01164066,"about_ca_system_score_codex":0.003040436,"about_ca_system_score_gemma":0.004234673,"threshold_uncertainty_score":0.05279535},"labels":[],"label_agreement":null},{"id":"W2909649608","doi":"","title":"Upper bounds on the sizes of variable strength covering arrays using the Lov\\'{a}sz local lemma","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of Ottawa","funders":"","keywords":"Hypergraph; Mathematics; Lemma (botany); Combinatorics; Upper and lower bounds; Logarithm; Generalization; Clique; Variable (mathematics); Row; Discrete mathematics; Spiral (railway); Logarithmic spiral; Algorithm; Computer science; Geometry; Mathematical analysis","score_opus":0.0798147617837113,"score_gpt":0.20125222415357577,"score_spread":0.12143746236986447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2909649608","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060967773,0.00086883287,0.9225208,0.0013588675,0.00006129754,0.00014382308,0.00051996234,0.0015274764,0.012031146],"genre_scores_gemma":[0.71623737,0.0014057428,0.27436483,0.0008148655,0.0003076662,0.000804062,0.0011276212,0.0011180065,0.0038198156],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99328315,0.0020840606,0.00035836807,0.0012760826,0.0022297488,0.00076848594],"domain_scores_gemma":[0.90767413,0.07623205,0.0045963908,0.0074664513,0.0026384783,0.0013925859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004751284,0.0016863549,0.0014347606,0.0022459882,0.0009149841,0.0027919791,0.002551541,0.0015060524,0.006371783],"category_scores_gemma":[0.04467228,0.001262983,0.0018636545,0.0026050222,0.00341481,0.007707046,0.0045580454,0.0037189585,0.0013954644],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011355303,0.00037901071,0.011811439,0.0010100356,0.00022587433,0.00059687294,0.00056813646,0.370737,0.063199565,0.35242772,0.013214836,0.18469396],"study_design_scores_gemma":[0.00010589615,0.0004641867,0.0028945347,0.00012219997,0.00016672279,0.0007217534,0.00012886447,0.6864107,0.03243735,0.26934046,0.007113655,0.00009370639],"about_ca_topic_score_codex":0.0008127599,"about_ca_topic_score_gemma":0.0010948067,"teacher_disagreement_score":0.006371783,"about_ca_system_score_codex":0.0018245599,"about_ca_system_score_gemma":0.0013970993,"threshold_uncertainty_score":0.02512747},"labels":[],"label_agreement":null},{"id":"W2912088669","doi":"","title":"Proceedings of the 6th International Workshop on Constraints in Software Testing, Verification, and Analysis","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Software testing; Computer science; Software engineering; Software verification; Software; Systems engineering; Programming language; Software construction; Engineering; Software development","score_opus":0.021075756774194165,"score_gpt":0.2534182041734346,"score_spread":0.23234244739924043,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912088669","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015496996,0.019809397,0.8981317,0.006606016,0.007433622,0.000404282,0.0006231255,0.0012855685,0.050209265],"genre_scores_gemma":[0.15855107,0.017076313,0.7169336,0.001975968,0.0047632204,0.00047847486,0.0036351485,0.001960189,0.09462593],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9948251,0.0023502365,0.00039909806,0.00074344163,0.0012722008,0.00040984142],"domain_scores_gemma":[0.98669577,0.007122793,0.00031277045,0.0026645379,0.002441783,0.00076239055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008344056,0.0016096039,0.0021426466,0.0018821493,0.001086675,0.0059445426,0.0029507086,0.0021213354,0.026202116],"category_scores_gemma":[0.012624724,0.0011775055,0.0017920092,0.001370777,0.0020362844,0.004600628,0.0029214185,0.0047008237,0.0041189874],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013130996,0.0010938125,0.0022247785,0.0010661831,0.00044642697,0.00080636214,0.0010295005,0.016632244,0.012385434,0.0983644,0.15916002,0.7054778],"study_design_scores_gemma":[0.00031949993,0.0005363484,0.004084723,0.0013811688,0.0004963637,0.0018266601,0.0006010929,0.17097044,0.025619317,0.18846257,0.60551196,0.00018987583],"about_ca_topic_score_codex":0.0052809655,"about_ca_topic_score_gemma":0.009505603,"teacher_disagreement_score":0.026202116,"about_ca_system_score_codex":0.0017582154,"about_ca_system_score_gemma":0.0031586902,"threshold_uncertainty_score":0.08765477},"labels":[],"label_agreement":null},{"id":"W2912170833","doi":"10.1002/spe.404","title":"Exploiting exceptions","year":2001,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Purdue University","keywords":"Bytecode; Java bytecode; Programming language; Computer science; Java; Compiler; Program optimization; Interpretation (philosophy); Optimizing compiler; Abstract interpretation; Code (set theory); Parallel computing; Operating system; Real time Java; Java annotation","score_opus":0.0296799226284058,"score_gpt":0.3130142890524758,"score_spread":0.28333436642407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912170833","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07944697,0.0010351072,0.8818221,0.00081471127,0.00027958254,0.00015842474,0.00021175729,0.020073744,0.01615763],"genre_scores_gemma":[0.6968661,0.0005989452,0.28274488,0.0007258921,0.0002010028,0.00013893982,0.00053532905,0.004160653,0.0140281515],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99575484,0.00096603704,0.000321295,0.0006133318,0.0018093203,0.0005352461],"domain_scores_gemma":[0.99245787,0.0028118105,0.00089749927,0.0026029684,0.0010544119,0.0001754749],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013665265,0.0011918491,0.00096188165,0.0009914511,0.000695202,0.0024317768,0.002106135,0.0008164856,0.004104174],"category_scores_gemma":[0.0074797673,0.00061775884,0.0008745177,0.00074895914,0.0017998043,0.0041073207,0.0024156154,0.0018540262,0.0015595566],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011727038,0.00029283253,0.0091408035,0.0009840198,0.00016623421,0.002040548,0.0012489143,0.052275896,0.13904105,0.16068155,0.020095542,0.61285996],"study_design_scores_gemma":[0.00020014627,0.0005477981,0.0023108746,0.00034382104,0.0004910836,0.0023755326,0.00026219038,0.31080863,0.3318938,0.21481064,0.13573135,0.00022415725],"about_ca_topic_score_codex":0.0012696038,"about_ca_topic_score_gemma":0.0012612441,"teacher_disagreement_score":0.004104174,"about_ca_system_score_codex":0.0005836678,"about_ca_system_score_gemma":0.0014660932,"threshold_uncertainty_score":0.013729811},"labels":[],"label_agreement":null},{"id":"W2912187704","doi":"10.1007/3-540-45324-5_109","title":"Humboldt Heroes","year":2001,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; League; Quarter (Canadian coin); Code (set theory); Operations research; Programming language; History; Engineering; Archaeology; Physics","score_opus":0.023497611520950187,"score_gpt":0.2601918604175893,"score_spread":0.2366942488966391,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912187704","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0025966195,0.028753895,0.004328391,0.0096932305,0.007287763,0.000024502902,0.0003063738,0.00037960682,0.94662964],"genre_scores_gemma":[0.009435819,0.006491409,0.0010072325,0.0008480276,0.0006327594,0.000015044648,0.00008194597,0.0001612648,0.98132646],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9997507,0.000042087275,0.000006487913,0.000055956443,0.000107440304,0.00003748035],"domain_scores_gemma":[0.99983084,0.00003482741,0.000012721073,0.000025681618,0.000048850503,0.000046964553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003194389,0.0009165516,0.00041475653,0.0011895286,0.0015598766,0.0022020652,0.00041391936,0.00097180257,0.07492513],"category_scores_gemma":[0.00078217324,0.0002792831,0.00021763219,0.00078847405,0.0011596915,0.0020453092,0.0019775012,0.0020039303,0.05167219],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003715323,0.000025436388,0.0001419686,0.00010475379,0.0000053255976,0.00013888716,0.00058641727,0.00027566543,0.00048816213,0.27451703,0.54756653,0.17611265],"study_design_scores_gemma":[0.0000010883074,0.000004190494,0.00004721573,0.000038108177,8.119617e-7,0.00006580893,0.00003383743,0.000034874876,0.00013324435,0.008757466,0.99088144,0.0000019571382],"about_ca_topic_score_codex":0.0007502886,"about_ca_topic_score_gemma":0.0018674762,"teacher_disagreement_score":0.07492513,"about_ca_system_score_codex":0.0009896184,"about_ca_system_score_gemma":0.00075407757,"threshold_uncertainty_score":0.2506495},"labels":[],"label_agreement":null},{"id":"W2913009888","doi":"","title":"Proceedings of the 7th International Workshop on Automation of Software Test","year":2012,"lang":"en","type":"article","venue":"International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Automation; Theme (computing); Scope (computer science); Test (biology); Software engineering; Software testing; Computer science; Selection (genetic algorithm); Engineering management; Diversity (politics); Software; Engineering; World Wide Web; Operating system; Programming language; Political science; Artificial intelligence","score_opus":0.030262160215990375,"score_gpt":0.26951956315226727,"score_spread":0.2392574029362769,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2913009888","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01298551,0.017828085,0.81635994,0.0074610338,0.0145754935,0.00080701004,0.0011111458,0.007366322,0.12150542],"genre_scores_gemma":[0.18851483,0.017857468,0.5540974,0.0033356454,0.005886529,0.0012624469,0.010621715,0.0035548063,0.21486916],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9938036,0.0022708343,0.0005813238,0.0007706671,0.002112647,0.00046090243],"domain_scores_gemma":[0.99156755,0.0029948512,0.00030518605,0.0021406512,0.0023328373,0.0006588594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007225926,0.0015154883,0.0014190247,0.0019359188,0.0009005338,0.0050622555,0.0024443236,0.001848923,0.038649235],"category_scores_gemma":[0.011286456,0.0007389213,0.0014312475,0.0011890496,0.0013466964,0.0035424968,0.0029921718,0.0039269435,0.011634962],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005468388,0.0003950695,0.0017246179,0.0005357271,0.00014787566,0.0008473963,0.0008436794,0.0052175024,0.009565905,0.038560368,0.20721939,0.7343957],"study_design_scores_gemma":[0.00009027244,0.00043328296,0.002768578,0.00060808135,0.00009668854,0.001838907,0.00038380464,0.02555076,0.010770163,0.04596836,0.91142184,0.0000693377],"about_ca_topic_score_codex":0.0014459726,"about_ca_topic_score_gemma":0.0022436,"teacher_disagreement_score":0.038649235,"about_ca_system_score_codex":0.00082962395,"about_ca_system_score_gemma":0.0018784299,"threshold_uncertainty_score":0.12929457},"labels":[],"label_agreement":null},{"id":"W2913486937","doi":"","title":"Proceedings of the 19th IFIP TC6/WG6.1 international conference, and 7th international conference on Testing of Software and Communicating Systems","year":2007,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Software; Computer science; Library science; Software engineering; Engineering; Operating system","score_opus":0.07168205796549357,"score_gpt":0.29696386027347144,"score_spread":0.22528180230797787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2913486937","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04804008,0.036169596,0.6610869,0.02651549,0.035475146,0.001625853,0.0032033012,0.004417949,0.18346563],"genre_scores_gemma":[0.18970121,0.03196359,0.2509123,0.003574587,0.01262809,0.0007060383,0.013318968,0.0037367563,0.49345845],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9971391,0.0012681867,0.00015158107,0.00023628862,0.0009690097,0.00023582802],"domain_scores_gemma":[0.99088645,0.0036935476,0.00018004682,0.0013701101,0.0029061423,0.0009637272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006175345,0.0012016086,0.0011660446,0.0017338993,0.0011998547,0.0040110513,0.0015738641,0.001955961,0.03088255],"category_scores_gemma":[0.0075556925,0.00062987953,0.0009655084,0.0011822595,0.0020006914,0.002479926,0.0015653118,0.0029077288,0.009585638],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001060907,0.0009916591,0.0047963606,0.0005212518,0.00020609461,0.0008633894,0.0008267979,0.005875338,0.018729346,0.015888128,0.45228052,0.49796015],"study_design_scores_gemma":[0.00017778763,0.0006457414,0.0074499804,0.00066687755,0.0002742985,0.0018890387,0.0006145814,0.025964899,0.032603014,0.035185512,0.8943998,0.00012847937],"about_ca_topic_score_codex":0.011069096,"about_ca_topic_score_gemma":0.02131779,"teacher_disagreement_score":0.03088255,"about_ca_system_score_codex":0.0017221229,"about_ca_system_score_gemma":0.002651474,"threshold_uncertainty_score":0.10331237},"labels":[],"label_agreement":null},{"id":"W2913811733","doi":"","title":"Proceedings of the Eighth International Workshop on Search-Based Software Testing","year":2015,"lang":"en","type":"article","venue":"International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Search-based software engineering; Software engineering; Computer science; Software development; Software metric; Software construction; Process (computing); Software; Metric (unit); Systems engineering; Engineering; Programming language","score_opus":0.09293096580281542,"score_gpt":0.2918747485365626,"score_spread":0.1989437827337472,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2913811733","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015917603,0.0418296,0.7496605,0.020680545,0.026259275,0.00071179465,0.0010196123,0.0036041567,0.14031696],"genre_scores_gemma":[0.16394666,0.035002835,0.47654903,0.005041243,0.0117732575,0.0011608645,0.008015517,0.0034841765,0.29502636],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9951267,0.0018245988,0.00040674716,0.00080355996,0.0014135326,0.0004247503],"domain_scores_gemma":[0.99124044,0.0042460673,0.00021325302,0.001517734,0.0020504647,0.0007320161],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0071771396,0.0018505695,0.0017991482,0.002052531,0.0009534154,0.005014424,0.0034266717,0.002555917,0.034709454],"category_scores_gemma":[0.013448485,0.00075112196,0.0017815239,0.0015778908,0.0020387997,0.0046852496,0.0030596836,0.00454166,0.008766591],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006064481,0.0005906315,0.001201514,0.0006070105,0.00019513113,0.0005939127,0.0006820934,0.017359078,0.005754842,0.06562207,0.2717212,0.63506603],"study_design_scores_gemma":[0.0001361388,0.00042415943,0.0015587662,0.00085707253,0.00012316277,0.00096709415,0.00039460242,0.054758217,0.0060078804,0.079410605,0.85526615,0.00009605098],"about_ca_topic_score_codex":0.0031028634,"about_ca_topic_score_gemma":0.0037754492,"teacher_disagreement_score":0.034709454,"about_ca_system_score_codex":0.0018108626,"about_ca_system_score_gemma":0.0023328874,"threshold_uncertainty_score":0.116114676},"labels":[],"label_agreement":null},{"id":"W2913863212","doi":"10.1007/s11219-019-9440-3","title":"Fault model-driven testing from FSM with symbolic inputs","year":2019,"lang":"en","type":"article","venue":"Software Quality Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Finite-state machine; Test suite; Computer science; Symbolic execution; Implementation; Fault model; Fault injection; Fault (geology); Set (abstract data type); Model-based testing; Theoretical computer science; Fault coverage; Process (computing); Constraint (computer-aided design); Algorithm; Programming language; Test case; Software; Engineering; Machine learning","score_opus":0.044427028615976964,"score_gpt":0.29178708709886186,"score_spread":0.2473600584828849,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2913863212","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06583197,0.00009717899,0.9273753,0.00018220332,0.0000414407,0.000087534405,0.00027229945,0.0035991156,0.0025129616],"genre_scores_gemma":[0.88411176,0.000053073283,0.11372559,0.00007422734,0.000017973947,0.00011033166,0.00036562167,0.00027941624,0.001261923],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99827266,0.00047440734,0.00006605992,0.00019464477,0.00082975166,0.00016236525],"domain_scores_gemma":[0.9922501,0.006197876,0.00032577902,0.0006676059,0.00049315847,0.00006562087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009337906,0.0010198068,0.00076089206,0.0012700465,0.0003353108,0.0007388509,0.0011179652,0.0009907731,0.003448348],"category_scores_gemma":[0.010744768,0.00033508873,0.00097275525,0.00067433,0.0010663408,0.0013799439,0.0012537916,0.0010940706,0.00035895436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051366753,0.00017310945,0.002097078,0.0003982434,0.0000826783,0.0009435636,0.00027805756,0.8015964,0.02530546,0.05358531,0.0014786496,0.1135477],"study_design_scores_gemma":[0.000026525111,0.000052555944,0.00019791232,0.000022623764,0.000016010783,0.00007868625,0.000010771173,0.9637212,0.009146635,0.026365839,0.0003536806,0.0000075839184],"about_ca_topic_score_codex":0.0027509637,"about_ca_topic_score_gemma":0.0040679793,"teacher_disagreement_score":0.003448348,"about_ca_system_score_codex":0.0010641209,"about_ca_system_score_gemma":0.0011415764,"threshold_uncertainty_score":0.011535823},"labels":[],"label_agreement":null},{"id":"W2916083331","doi":"","title":"Advances in Debug Automation for a Modern Verification Environment","year":2013,"lang":"en","type":"dissertation","venue":"TSpace (University of Toronto)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto","keywords":"Debugging; Automation; Software engineering; Computer science; Systems engineering; Engineering; Programming language","score_opus":0.012658475127772163,"score_gpt":0.24800232028868463,"score_spread":0.23534384516091247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2916083331","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030185198,0.002361177,0.9816861,0.0014655876,0.0002029992,0.00004807787,0.00006212116,0.0034565579,0.007698738],"genre_scores_gemma":[0.101973295,0.005376783,0.8833196,0.0007659759,0.00038698953,0.00012275563,0.00028926603,0.0011611686,0.0066041676],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99003905,0.00334995,0.00055987306,0.0016941341,0.0039417157,0.0004152832],"domain_scores_gemma":[0.98187673,0.008227987,0.00080494944,0.006866675,0.0019723342,0.00025126332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007070682,0.0012800918,0.00076330174,0.0022731556,0.0009355882,0.004097103,0.0026073016,0.0015900106,0.007618873],"category_scores_gemma":[0.018915646,0.0013265043,0.0016194268,0.0012978653,0.0023732134,0.010457869,0.003533323,0.0059117414,0.002628358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010960687,0.00007652728,0.001097723,0.0005066914,0.0000600611,0.00016718535,0.0007739738,0.020934975,0.011071436,0.47986656,0.0095220255,0.47581318],"study_design_scores_gemma":[0.0001033877,0.00022512044,0.0010249048,0.000722957,0.000104921724,0.000828431,0.0002554518,0.17532066,0.026238652,0.38661197,0.40842173,0.00014177508],"about_ca_topic_score_codex":0.0016248765,"about_ca_topic_score_gemma":0.0015868207,"teacher_disagreement_score":0.007618873,"about_ca_system_score_codex":0.0016795764,"about_ca_system_score_gemma":0.0028223512,"threshold_uncertainty_score":0.03739375},"labels":[],"label_agreement":null},{"id":"W2921594963","doi":"10.1007/s10664-019-09691-z","title":"Assessing and optimizing the performance impact of the just-in-time configuration parameters - a case study on PyPy","year":2019,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reliability engineering; Engineering","score_opus":0.03857829876151698,"score_gpt":0.31762281544977644,"score_spread":0.27904451668825947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2921594963","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99033254,0.000105015744,0.007749113,0.00007367966,0.000006365576,0.000051290426,0.00007270601,0.00041957092,0.0011896025],"genre_scores_gemma":[0.98853403,0.00005097857,0.010692519,0.000014520705,0.0000027658568,0.000021437894,0.00009173746,0.000094246345,0.00049781514],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964065,0.0018414428,0.0001988652,0.00045284693,0.00076498894,0.00033539],"domain_scores_gemma":[0.94614005,0.040937524,0.0026143226,0.0066106003,0.003008332,0.00068918435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044340324,0.0008703152,0.00050631317,0.0008845444,0.0005957846,0.0011196812,0.0018189315,0.0010827967,0.0014548635],"category_scores_gemma":[0.043176048,0.0003947959,0.0002868169,0.0011255222,0.00102032,0.0024033377,0.0009521174,0.0013077498,0.00030351346],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005317246,0.010456096,0.11693261,0.0012788558,0.00025709296,0.0016785525,0.0035455308,0.35591057,0.09464117,0.007138381,0.0030082264,0.39983565],"study_design_scores_gemma":[0.00046716176,0.008538543,0.111352555,0.0001523769,0.00029095498,0.0009997693,0.0024909815,0.7400948,0.12426668,0.0055607394,0.005591042,0.0001944356],"about_ca_topic_score_codex":0.0032454943,"about_ca_topic_score_gemma":0.0028493756,"teacher_disagreement_score":0.0044340324,"about_ca_system_score_codex":0.0008463407,"about_ca_system_score_gemma":0.0012143563,"threshold_uncertainty_score":0.023449719},"labels":[],"label_agreement":null},{"id":"W2922606767","doi":"10.1109/icpc.2019.00020","title":"An Empirical Study on Practicality of Specification Mining Algorithms on a Real-World Application","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Debugging; Computer science; Program comprehension; Context (archaeology); Implementation; Inference; Programming language; Abstraction; Set (abstract data type); Software engineering; Root cause; Software; Algorithm; Artificial intelligence; Software system; Reliability engineering","score_opus":0.12292230898323768,"score_gpt":0.42842618468842586,"score_spread":0.30550387570518817,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2922606767","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98279506,0.0011537258,0.010881744,0.0007636989,0.000035358386,0.00019705419,0.00081596733,0.0004790943,0.0028783502],"genre_scores_gemma":[0.9838213,0.00028161675,0.013301995,0.00007630926,0.00002736199,0.00011345741,0.0016993969,0.00012907317,0.0005495842],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97210896,0.015314953,0.0027818892,0.0037067386,0.0053383573,0.0007490599],"domain_scores_gemma":[0.4286365,0.5090995,0.013611459,0.030741489,0.015737096,0.0021740773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027885797,0.0009369105,0.00068612106,0.0028758256,0.0008943356,0.0023975736,0.0020367769,0.0020530468,0.0021873084],"category_scores_gemma":[0.32447308,0.00057586975,0.0010309647,0.0038585002,0.0022761035,0.0055036754,0.0016978177,0.0032418591,0.0009768747],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0048692115,0.006894337,0.5054127,0.0024547516,0.00093298027,0.0007687566,0.0056451745,0.100426674,0.008676514,0.007035826,0.011144484,0.34573868],"study_design_scores_gemma":[0.00072732946,0.0062552197,0.31692502,0.00044778452,0.0004765314,0.0025069937,0.0052909045,0.6197029,0.016100125,0.011701792,0.019662699,0.00020273359],"about_ca_topic_score_codex":0.0016370261,"about_ca_topic_score_gemma":0.0017516353,"teacher_disagreement_score":0.027885797,"about_ca_system_score_codex":0.0014574176,"about_ca_system_score_gemma":0.00093651126,"threshold_uncertainty_score":0.14747596},"labels":[],"label_agreement":null},{"id":"W2925127136","doi":"10.1007/s11219-018-9437-3","title":"Testing self-healing cyber-physical systems under uncertainty: a fragility-oriented approach","year":2019,"lang":"en","type":"article","venue":"Software Quality Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Horizon 2020 Framework Programme; Norges Forskningsråd","keywords":"Fragility; Reliability engineering; Normality; Computer science; Reliability (semiconductor); Cyber-physical system; Data mining; Engineering; Mathematics; Statistics","score_opus":0.04131213216160864,"score_gpt":0.308153895772097,"score_spread":0.2668417636104884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2925127136","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030535502,0.0002591953,0.9653995,0.0007159987,0.000031532243,0.00008192324,0.00003354599,0.0002887724,0.0026541029],"genre_scores_gemma":[0.85621095,0.00034492958,0.14156246,0.0002353281,0.00008299107,0.00013495982,0.000059809063,0.0000936826,0.0012747884],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965875,0.0014807327,0.00013919758,0.00038618344,0.0010867787,0.0003195131],"domain_scores_gemma":[0.9823162,0.013864918,0.0012994022,0.0011531773,0.0010794827,0.00028690003],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047735344,0.0012648224,0.0012673995,0.0022469715,0.0007685533,0.0017754844,0.0025451365,0.001583158,0.0015324156],"category_scores_gemma":[0.01935368,0.0005434034,0.0013974147,0.0009732832,0.003906408,0.0031881293,0.0025541696,0.001810398,0.00012154525],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016207468,0.00014326042,0.0030979582,0.0002655426,0.00018235252,0.0004249565,0.0004655585,0.7956812,0.004424917,0.13741727,0.00057815167,0.057156794],"study_design_scores_gemma":[0.000012506993,0.00007731205,0.00026855065,0.000032868116,0.000040196348,0.00007704707,0.00009407955,0.9163173,0.0016120455,0.08112196,0.00033227136,0.000013904977],"about_ca_topic_score_codex":0.0025210597,"about_ca_topic_score_gemma":0.0017847456,"teacher_disagreement_score":0.0047735344,"about_ca_system_score_codex":0.0012433336,"about_ca_system_score_gemma":0.0014662859,"threshold_uncertainty_score":0.02524513},"labels":[],"label_agreement":null},{"id":"W2930178786","doi":"","title":"PySnippet: Accelerating Exploratory Data Analysis in Jupyter Notebook through Facilitated Access to Example Code.","year":2019,"lang":"en","type":"article","venue":"EDBT/ICDT Workshops","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Code (set theory); Exploratory analysis; Computer security; Exploratory data analysis; Programming language; Data science; Data mining","score_opus":0.17585169340228715,"score_gpt":0.3512547445601635,"score_spread":0.17540305115787633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2930178786","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009511881,0.0002743772,0.17725831,0.00081572885,0.00063827744,0.00086446863,0.083979264,0.7156361,0.011021622],"genre_scores_gemma":[0.07336773,0.00068153354,0.55384386,0.0012017637,0.00028931038,0.0036040607,0.118628465,0.20992815,0.038455166],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989735,0.00019176291,0.000083967134,0.00020475601,0.0004504121,0.00009569195],"domain_scores_gemma":[0.99057174,0.006518507,0.00032875678,0.0012267504,0.00096279837,0.00039149178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025077453,0.0020850147,0.00093196484,0.0021261626,0.00078479864,0.0018305978,0.002283319,0.00091540004,0.15230472],"category_scores_gemma":[0.016220486,0.0010293827,0.0009388958,0.0015901875,0.00050674943,0.0026626836,0.0029710706,0.002605073,0.035577554],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009472857,0.00023039579,0.0027501662,0.0013685992,0.00014518038,0.0006429191,0.00088974234,0.0014715968,0.011163846,0.0027480822,0.84160155,0.13604072],"study_design_scores_gemma":[0.0017989669,0.00048043876,0.0231196,0.0014410551,0.000244475,0.0011071387,0.0009151473,0.10371984,0.09607996,0.028386952,0.7422269,0.00047959774],"about_ca_topic_score_codex":0.0031874354,"about_ca_topic_score_gemma":0.0070384094,"teacher_disagreement_score":0.15230472,"about_ca_system_score_codex":0.0004659389,"about_ca_system_score_gemma":0.001989598,"threshold_uncertainty_score":0.50951004},"labels":[],"label_agreement":null},{"id":"W2932065239","doi":"10.1007/978-3-030-16722-6_24","title":": Priority Aware Test Case Reduction","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reduction (mathematics); Test case; Debugging; Code coverage; Test (biology); Abstract syntax tree; Programming language; Queue; Priority queue; Process (computing); Set (abstract data type); Software; Reliability engineering; Machine learning; Parsing","score_opus":0.020987685690595698,"score_gpt":0.26446983885645753,"score_spread":0.24348215316586183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2932065239","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029225672,0.0013763416,0.9298493,0.00064690894,0.00029309368,0.00048533472,0.0005803767,0.02374992,0.0137930075],"genre_scores_gemma":[0.23208763,0.0007967255,0.74076366,0.0007542123,0.0001935074,0.0003737357,0.003262975,0.0036935024,0.018073985],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99546707,0.0006933442,0.00023619473,0.0006290932,0.0026519632,0.00032234646],"domain_scores_gemma":[0.9938969,0.0019790998,0.0004567732,0.0020710032,0.0014432048,0.00015308656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015940131,0.0015246731,0.0006540567,0.0022134779,0.0004946539,0.0017679681,0.0028978938,0.0009194775,0.0055036913],"category_scores_gemma":[0.008277807,0.0006908097,0.0012154191,0.0017727114,0.0008386719,0.0022361279,0.0016514261,0.0020955598,0.002885454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029003393,0.0002955471,0.0021807733,0.0004409114,0.00006865862,0.0003947863,0.0002665575,0.016318938,0.062426064,0.015461418,0.026458465,0.8753979],"study_design_scores_gemma":[0.00026523464,0.0008107933,0.0059943805,0.00021405963,0.00025587133,0.003475079,0.00031712992,0.48314852,0.29356325,0.07467181,0.13711123,0.00017275219],"about_ca_topic_score_codex":0.0025718475,"about_ca_topic_score_gemma":0.0030757103,"teacher_disagreement_score":0.0055036913,"about_ca_system_score_codex":0.00097598345,"about_ca_system_score_gemma":0.0020876704,"threshold_uncertainty_score":0.018411636},"labels":[],"label_agreement":null},{"id":"W2942132945","doi":"10.1007/978-3-030-10543-3_13","title":"Evaluation and Application of Two Fuzzing Approaches for Security Testing of IoT Applications","year":2019,"lang":"en","type":"book-chapter","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Fuzz testing; Computer science; Software quality; Software security assurance; Software bug; Software; Software reliability testing; Software quality assurance; Software metric; Security testing; Code coverage; Leverage (statistics); Software testing; Reliability engineering; Software engineering; Software development; Computer security; Machine learning; Operating system; Engineering; Information security","score_opus":0.10769492292998101,"score_gpt":0.318746074156001,"score_spread":0.21105115122601997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2942132945","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44768605,0.0012537724,0.53710556,0.0006037154,0.00015354005,0.00054610474,0.00030397816,0.0036525223,0.00869476],"genre_scores_gemma":[0.7405641,0.0002535104,0.25656196,0.00010209194,0.000019639147,0.00014083245,0.00023043707,0.0001628923,0.0019645903],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99591607,0.0012305469,0.0002087568,0.0004834279,0.001933279,0.00022801045],"domain_scores_gemma":[0.9829707,0.012085799,0.00087138615,0.0017776269,0.002006948,0.00028755373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003518005,0.0011429671,0.0005310605,0.0022338214,0.0004811601,0.0013917808,0.0018455796,0.0011980445,0.0025092377],"category_scores_gemma":[0.017466968,0.00032331224,0.0006313939,0.0009520602,0.0011988482,0.0020130735,0.0010290493,0.0011030696,0.00020481912],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027262932,0.0016670671,0.0154092815,0.0008505983,0.00029199847,0.00029528767,0.0010366645,0.1605968,0.10491153,0.036055066,0.0026089507,0.67355037],"study_design_scores_gemma":[0.0001823844,0.0013742999,0.005738359,0.000115702605,0.00026956823,0.0003495399,0.0002994046,0.8847993,0.092593096,0.011607053,0.0026032142,0.00006806543],"about_ca_topic_score_codex":0.0036733272,"about_ca_topic_score_gemma":0.0049695796,"teacher_disagreement_score":0.0036733272,"about_ca_system_score_codex":0.001696463,"about_ca_system_score_gemma":0.0012432656,"threshold_uncertainty_score":0.018605173},"labels":[],"label_agreement":null},{"id":"W2948233389","doi":"10.1007/978-3-030-27455-9_10","title":"Revisiting Hyper-Parameter Tuning for Search-Based Test Data Generation","year":2019,"lang":"en","type":"preprint","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Metric (unit); Replicate; Heuristic; Computer science; Baseline (sea); Software; Performance metric; Minor (academic); Space (punctuation); Machine learning; Artificial intelligence; Mathematics; Statistics; Engineering","score_opus":0.10774043282988424,"score_gpt":0.338558522507848,"score_spread":0.23081808967796377,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2948233389","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03327697,0.00093073666,0.9329136,0.0005537813,0.00019545728,0.00031989545,0.00025865668,0.026773255,0.004777642],"genre_scores_gemma":[0.49184468,0.00024925297,0.49895766,0.00074492185,0.00009348462,0.0002643942,0.00059094303,0.004102329,0.0031523558],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98998356,0.004217665,0.00081723265,0.0015249021,0.0027736465,0.00068298884],"domain_scores_gemma":[0.9662212,0.01949074,0.0010128261,0.009380869,0.0033461088,0.00054826617],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050808038,0.0016862337,0.001728198,0.00227882,0.00073331635,0.0028920248,0.004652451,0.0021502615,0.010011042],"category_scores_gemma":[0.04016911,0.0010826241,0.0012053783,0.0015543995,0.0015181878,0.0039811446,0.0037682464,0.002721552,0.0035554627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013159324,0.0005091579,0.005792066,0.0006070493,0.0002171187,0.00035133786,0.0005603531,0.08778561,0.04274691,0.011842009,0.009841297,0.8384311],"study_design_scores_gemma":[0.00022172034,0.0003037314,0.0011644114,0.00020340596,0.00012733032,0.00034890688,0.00012956215,0.9276706,0.038852032,0.02199062,0.008911965,0.0000757022],"about_ca_topic_score_codex":0.004134506,"about_ca_topic_score_gemma":0.0051370687,"teacher_disagreement_score":0.010011042,"about_ca_system_score_codex":0.0013706407,"about_ca_system_score_gemma":0.0029356705,"threshold_uncertainty_score":0.0334903},"labels":[],"label_agreement":null},{"id":"W2950434324","doi":"10.82308/1356","title":"On testing concurrent systems through contexts of queues","year":2006,"lang":"en","type":"article","venue":"eScholarship@McGill (McGill)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fonds Québécois de la Recherche sur la Nature et les Technologies; Natural Sciences and Engineering Research Council of Canada; McGill University; Microsoft Research","keywords":"Asynchronous communication; Computer science; Queue; Atomicity; Fork–join queue; Distributed computing; Queue management system; Programming language; Computer network","score_opus":0.032207299057107355,"score_gpt":0.2550858479032864,"score_spread":0.22287854884617908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950434324","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10773757,0.0016434771,0.88105947,0.0009984443,0.000054631866,0.00015886012,0.000034559882,0.00046407027,0.0078488765],"genre_scores_gemma":[0.5961823,0.002621547,0.39854616,0.00047540513,0.00011202556,0.0002776181,0.00009367855,0.00014902599,0.001542225],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9944864,0.0031601195,0.00032121607,0.0005271282,0.0010904138,0.0004147221],"domain_scores_gemma":[0.97781914,0.018851766,0.00072307076,0.0013687023,0.0009991165,0.00023828067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043876055,0.0009177961,0.0006589461,0.001578437,0.000854859,0.0013720478,0.0012978826,0.0009207688,0.0010500532],"category_scores_gemma":[0.017825749,0.0005104488,0.0009858392,0.0014588633,0.0039813207,0.004575758,0.0028293005,0.0014574717,0.00016521174],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002460587,0.00018686849,0.005306249,0.00041510176,0.0000503711,0.00063332834,0.0020949137,0.13500245,0.014310497,0.63008165,0.0007587785,0.21091373],"study_design_scores_gemma":[0.0001041914,0.0008287386,0.0019666385,0.0005938445,0.00014967275,0.00077641325,0.00066561427,0.3715125,0.026636787,0.58529145,0.011358525,0.000115647556],"about_ca_topic_score_codex":0.0028154415,"about_ca_topic_score_gemma":0.0020710505,"teacher_disagreement_score":0.0043876055,"about_ca_system_score_codex":0.0009012912,"about_ca_system_score_gemma":0.0015313159,"threshold_uncertainty_score":0.023204148},"labels":[],"label_agreement":null},{"id":"W2951101154","doi":"10.1109/icstw.2019.00029","title":"Using Imprecise Test Oracles Modelled by FSM","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Oracle; Nondeterministic algorithm; Computer science; Context (archaeology); Finite-state machine; Conformance testing; Set (abstract data type); Test case; Theoretical computer science; Algorithm; Machine learning; Programming language","score_opus":0.027730293977826826,"score_gpt":0.27313344330657946,"score_spread":0.24540314932875262,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2951101154","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012967806,0.00022413924,0.98357403,0.00018593053,0.000024990035,0.00008547076,0.0002070365,0.001090171,0.0016403768],"genre_scores_gemma":[0.5514288,0.00046391154,0.44395816,0.00022779214,0.00008055156,0.00032650496,0.00068351853,0.00030471233,0.0025260744],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98664397,0.0046620313,0.0011109221,0.002106574,0.0046310853,0.0008455053],"domain_scores_gemma":[0.9768847,0.015683832,0.0014451476,0.004191003,0.0014713433,0.00032386289],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070935367,0.0017427099,0.0013496039,0.0025506115,0.0006323954,0.0040471912,0.0034025041,0.0026995763,0.0032792434],"category_scores_gemma":[0.03167782,0.00096645363,0.0021971234,0.0014527274,0.0045886706,0.007108871,0.00261265,0.00302401,0.00074336736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034838074,0.00007119302,0.0024586318,0.0002692142,0.00009404885,0.0007998064,0.0005817367,0.7082978,0.006056352,0.2326365,0.00087643985,0.047509935],"study_design_scores_gemma":[0.000045705092,0.00008330214,0.00024819787,0.000072115785,0.00004674061,0.00022119241,0.000043066535,0.81682926,0.00669611,0.17161353,0.004060986,0.000039731774],"about_ca_topic_score_codex":0.0046473066,"about_ca_topic_score_gemma":0.0034674893,"teacher_disagreement_score":0.0070935367,"about_ca_system_score_codex":0.0020786636,"about_ca_system_score_gemma":0.0015032586,"threshold_uncertainty_score":0.037514627},"labels":[],"label_agreement":null},{"id":"W2952903800","doi":"10.1109/icst.2019.00019","title":"BugsJS: a Benchmark of JavaScript Bugs","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":94,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"JavaScript; Unobtrusive JavaScript; Computer science; Benchmark (surveying); Programming language; Test case; Software bug; Unit testing; Rich Internet application; Operating system; Database; Software; Machine learning","score_opus":0.004992930306730839,"score_gpt":0.20213004230445228,"score_spread":0.19713711199772144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952903800","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7875907,0.009624546,0.08969591,0.0010486011,0.0006564929,0.0013628403,0.03692102,0.059616737,0.01348318],"genre_scores_gemma":[0.70348644,0.0022388743,0.17264806,0.0006038807,0.00014631366,0.00108743,0.10710485,0.008819089,0.003865],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9892736,0.002322168,0.0015139044,0.0014977916,0.004823578,0.0005689814],"domain_scores_gemma":[0.967399,0.014963796,0.0040583597,0.00504528,0.0072895004,0.0012441475],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056681396,0.0016642379,0.00064910407,0.005360328,0.00067923183,0.0011414157,0.0029044596,0.0012727785,0.0010667143],"category_scores_gemma":[0.034399055,0.0005437106,0.00094413676,0.0042339447,0.0011328813,0.001815648,0.0017117701,0.0014148746,0.00092955807],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002865691,0.0032973725,0.19901717,0.014334993,0.0012777469,0.0026023102,0.0036166043,0.078350104,0.0882781,0.010705129,0.16648026,0.42917454],"study_design_scores_gemma":[0.001067933,0.004410102,0.25176346,0.001921781,0.00083729665,0.0051225284,0.0021129062,0.39288136,0.13207808,0.015243836,0.19200654,0.0005542094],"about_ca_topic_score_codex":0.004403885,"about_ca_topic_score_gemma":0.0062788217,"teacher_disagreement_score":0.0056681396,"about_ca_system_score_codex":0.0007938062,"about_ca_system_score_gemma":0.001774942,"threshold_uncertainty_score":0.029976308},"labels":[],"label_agreement":null},{"id":"W2953988581","doi":"10.1109/icse-companion.2019.00095","title":"Supervised Tie Breaking in Test Case Prioritization","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Prioritization; Test (biology); Test case; Code coverage; Rank (graph theory); Data mining; Code (set theory); Cover (algebra); Reliability engineering; Machine learning; Software; Engineering; Mathematics; Programming language; Regression analysis","score_opus":0.011334825375838005,"score_gpt":0.2420150815609072,"score_spread":0.2306802561850692,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953988581","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13080171,0.0010179193,0.86118627,0.0005848056,0.00012419661,0.00051769865,0.00027786783,0.0033742534,0.0021152773],"genre_scores_gemma":[0.7922508,0.0002037597,0.20371011,0.00037065666,0.0001644405,0.00034182658,0.0010915202,0.00031126945,0.001555638],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9922347,0.0026735596,0.0005173944,0.0017050804,0.0021889552,0.0006803162],"domain_scores_gemma":[0.9500102,0.03515481,0.0051289583,0.003029831,0.0050518787,0.0016243178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007385337,0.0017521816,0.002496863,0.004458146,0.0010774892,0.0017552392,0.0027617903,0.0018207247,0.001777354],"category_scores_gemma":[0.041854244,0.0009433945,0.0010909998,0.0021510895,0.0011571242,0.0025582535,0.0020756994,0.0030356694,0.0005475093],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007307082,0.0008458202,0.02264401,0.00039924524,0.0002894228,0.00036045606,0.0005713693,0.45811144,0.009003551,0.0053663,0.0052677337,0.49640986],"study_design_scores_gemma":[0.000040180996,0.0001380156,0.001313647,0.000018906703,0.00003101218,0.00006950823,0.000043906195,0.98900384,0.0024827663,0.0061328104,0.0007118538,0.000013471661],"about_ca_topic_score_codex":0.00561353,"about_ca_topic_score_gemma":0.007768839,"teacher_disagreement_score":0.007385337,"about_ca_system_score_codex":0.0014813232,"about_ca_system_score_gemma":0.0030935083,"threshold_uncertainty_score":0.03905791},"labels":[],"label_agreement":null},{"id":"W2954796040","doi":"10.1109/tse.2019.2925345","title":"What's Wrong with My Benchmark Results? Studying Bad Practices in JMH Benchmarks","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Vetenskapsrådet","keywords":"Java; Computer science; Benchmark (surveying); Software engineering; Open source; Statement (logic); Best practice; Empirical research; Software; Programming language; Data science","score_opus":0.01793363133666667,"score_gpt":0.24993281964352593,"score_spread":0.23199918830685926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2954796040","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84823877,0.006395219,0.08478143,0.029764337,0.0017643078,0.0004449777,0.002255945,0.012397343,0.013957748],"genre_scores_gemma":[0.90653825,0.001374499,0.07916202,0.004560225,0.0003046464,0.00028524984,0.0017063181,0.003973196,0.002095669],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.94298553,0.016762555,0.0055161836,0.0064613787,0.025804428,0.002469903],"domain_scores_gemma":[0.72016096,0.13211912,0.038443673,0.04546117,0.059254583,0.0045604073],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.035781264,0.0014890275,0.001219866,0.0054103937,0.0020380071,0.005827317,0.0029314377,0.0019458486,0.0010280415],"category_scores_gemma":[0.25754267,0.0011182447,0.0010691839,0.006552599,0.0036007501,0.007134066,0.0025579275,0.003448797,0.0008943212],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079016475,0.00091602997,0.49943078,0.0017064677,0.0009916914,0.0012264898,0.012757837,0.015293237,0.022995869,0.009340058,0.052005358,0.38254604],"study_design_scores_gemma":[0.000367405,0.0031146395,0.5499659,0.004591341,0.0016915618,0.004978426,0.021066418,0.105749615,0.13830219,0.050796162,0.118205875,0.0011704547],"about_ca_topic_score_codex":0.006064502,"about_ca_topic_score_gemma":0.008025615,"teacher_disagreement_score":0.96421874,"about_ca_system_score_codex":0.0028411963,"about_ca_system_score_gemma":0.0021502164,"threshold_uncertainty_score":0.1892317},"labels":[],"label_agreement":null},{"id":"W2955338589","doi":"10.22215/etd/2018-13266","title":"FSM Testing Based on Transition Trees and Complete Round Trip Paths Testing Criteria","year":2018,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Tree traversal; Random testing; Test suite; Algorithm; Tree (set theory); Computer science; Finite-state machine; Test case; Fault tree analysis; Mathematics; Engineering; Machine learning; Combinatorics; Reliability engineering","score_opus":0.06657820508724317,"score_gpt":0.3033449460471787,"score_spread":0.23676674095993555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955338589","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12418754,0.0009982549,0.84915715,0.0006198501,0.00008015388,0.00036354587,0.0006445224,0.0015478876,0.022401115],"genre_scores_gemma":[0.78194773,0.0003427267,0.21397962,0.00014605513,0.000057909918,0.00022611434,0.00082910806,0.00021821068,0.0022525224],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99141675,0.002957979,0.00046809705,0.0009236688,0.0037957088,0.00043779344],"domain_scores_gemma":[0.9695322,0.02321488,0.0015636593,0.0020508468,0.0029754057,0.00066302327],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003245669,0.0007483028,0.00065221515,0.0028789889,0.00042247807,0.0015529734,0.0011225123,0.00087527215,0.0035691843],"category_scores_gemma":[0.0384653,0.00029426842,0.00088504475,0.0015759644,0.0017920731,0.0037100227,0.0013962967,0.0011231042,0.0005462731],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066870934,0.0003229673,0.009722262,0.0008175637,0.00017029895,0.00063171546,0.0007259581,0.22825062,0.0156180505,0.36727697,0.00574915,0.37004578],"study_design_scores_gemma":[0.00007364705,0.00093552546,0.0048549864,0.0003560627,0.00009416134,0.0009973006,0.0002497728,0.672647,0.016595079,0.29080892,0.012309225,0.00007830123],"about_ca_topic_score_codex":0.0016290512,"about_ca_topic_score_gemma":0.001806613,"teacher_disagreement_score":0.0035691843,"about_ca_system_score_codex":0.0011030476,"about_ca_system_score_gemma":0.0016923195,"threshold_uncertainty_score":0.017164946},"labels":[],"label_agreement":null},{"id":"W2955364532","doi":"10.1109/icse-seip.2019.00031","title":"Improving Test Effectiveness Using Test Executions History: An Industrial Experience Report","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Test (biology); Test Management Approach; Test strategy; Regression testing; Integration testing; Prioritization; Software engineering; Selection (genetic algorithm); System integration testing; Test case; Risk-based testing; Software; Reliability engineering; Software quality; Software development; Process management; Machine learning; Engineering; Software construction","score_opus":0.0682729781571386,"score_gpt":0.2945401461666829,"score_spread":0.22626716800954427,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955364532","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91422254,0.0013964503,0.06674636,0.0026372904,0.00006304132,0.00049680436,0.0003003526,0.0020305314,0.012106727],"genre_scores_gemma":[0.9238638,0.00092070736,0.07141325,0.000217326,0.000044257256,0.00009047719,0.00045719632,0.0002972733,0.0026955928],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98744905,0.007118584,0.0007293859,0.00081401324,0.0033251462,0.0005638007],"domain_scores_gemma":[0.93612015,0.041696932,0.0024279866,0.0053925496,0.012003488,0.0023589723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019538417,0.0009202943,0.0005044352,0.0018068228,0.0007248641,0.0020949657,0.0024944844,0.00096795196,0.001572784],"category_scores_gemma":[0.055902604,0.0004672432,0.00037219052,0.0014537077,0.0010789725,0.002784209,0.0014316181,0.0012737754,0.00052753394],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063373253,0.005894643,0.051646285,0.0004951659,0.00010639975,0.00075304625,0.012019016,0.020453578,0.015964711,0.002156802,0.010925363,0.8789512],"study_design_scores_gemma":[0.0013546129,0.052274097,0.20850079,0.0014615094,0.0010793676,0.007975401,0.02831004,0.34966964,0.20256697,0.008419185,0.13755344,0.0008349955],"about_ca_topic_score_codex":0.0038971805,"about_ca_topic_score_gemma":0.005384031,"teacher_disagreement_score":0.019538417,"about_ca_system_score_codex":0.00134194,"about_ca_system_score_gemma":0.0015074788,"threshold_uncertainty_score":0.103330255},"labels":[],"label_agreement":null},{"id":"W2955925687","doi":"10.1109/icse.2019.00031","title":"Mining Historical Test Logs to Predict Bugs and Localize Faults in the Test Logs","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":71,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Flagging; Fault (geology); Product (mathematics); Data mining; Test data; Software; Reliability engineering; Engineering; Mathematics; Software engineering","score_opus":0.016333578780329608,"score_gpt":0.2420664346322286,"score_spread":0.225732855851899,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955925687","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6593813,0.0030447966,0.26140797,0.0009219861,0.00023147558,0.00046472345,0.028330566,0.042824212,0.0033929637],"genre_scores_gemma":[0.85727376,0.00065616245,0.10585564,0.0001399564,0.00009033542,0.00020157549,0.0336836,0.0004835955,0.0016154376],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99891186,0.00017606036,0.00013444798,0.00026637746,0.00039421875,0.0001170397],"domain_scores_gemma":[0.98926276,0.0051442767,0.0015575914,0.0013931281,0.002200718,0.00044151832],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015138512,0.0015078146,0.00074423285,0.0073063946,0.00030508733,0.0009955975,0.0013535254,0.00085534796,0.0012030206],"category_scores_gemma":[0.010902539,0.0004292275,0.00070239324,0.0027998632,0.00028202546,0.0015942272,0.0004916359,0.00081939664,0.0011164667],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008548995,0.0010666299,0.2409158,0.00088683114,0.00030742443,0.0016709507,0.00034377736,0.13904318,0.019447487,0.0011668219,0.02098403,0.57331216],"study_design_scores_gemma":[0.000051695726,0.00039362427,0.04566069,0.000090194546,0.00011073591,0.00067847624,0.00023516237,0.92740303,0.015391323,0.00365709,0.006275391,0.00005248533],"about_ca_topic_score_codex":0.0071491054,"about_ca_topic_score_gemma":0.010760507,"teacher_disagreement_score":0.0073063946,"about_ca_system_score_codex":0.0005946842,"about_ca_system_score_gemma":0.00087436975,"threshold_uncertainty_score":0.0142149925},"labels":[],"label_agreement":null},{"id":"W2960484385","doi":"10.1145/3356773.3356801","title":"The State and Future of Genetic Improvement","year":2019,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Engineering and Physical Sciences Research Council","keywords":"Maintainability; Session (web analytics); State (computer science); Software; Software development","score_opus":0.006067583557949196,"score_gpt":0.20652276807165942,"score_spread":0.20045518451371022,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2960484385","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017236277,0.6038074,0.08991433,0.21387762,0.0038116476,0.00006234752,0.0002255197,0.00057369535,0.070491165],"genre_scores_gemma":[0.35082483,0.5006046,0.088088505,0.035073787,0.008761267,0.0003276089,0.00053071376,0.0003920387,0.015396606],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9938791,0.0031172931,0.00020438006,0.00088287255,0.0014914661,0.0004249071],"domain_scores_gemma":[0.96983004,0.023638014,0.000702733,0.0024653412,0.0025319292,0.0008319777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019599343,0.000689171,0.0012704555,0.0017158703,0.0011486018,0.0059304144,0.0024547135,0.0054655797,0.008190845],"category_scores_gemma":[0.027173286,0.0003105442,0.000839098,0.0021870316,0.011241385,0.012455665,0.0025894234,0.004793149,0.001444002],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028225937,0.00017017839,0.0011335997,0.0010666985,0.0000736603,0.00010931469,0.0008854394,0.007957859,0.000909805,0.52830803,0.019548366,0.4395548],"study_design_scores_gemma":[0.00007394703,0.00020590638,0.00079318264,0.0011349822,0.000036767793,0.0001585807,0.0005496959,0.007168289,0.0007456365,0.7096188,0.2794517,0.00006249875],"about_ca_topic_score_codex":0.0030096837,"about_ca_topic_score_gemma":0.0016513523,"teacher_disagreement_score":0.019599343,"about_ca_system_score_codex":0.0035231488,"about_ca_system_score_gemma":0.0036617422,"threshold_uncertainty_score":0.10365248},"labels":[],"label_agreement":null},{"id":"W2964374341","doi":"10.1007/978-3-030-28423-7_9","title":"Mutant Accuracy Testing for Assessing the Implementation of Numerical Algorithms","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of British Columbia","funders":"","keywords":"Implementation; Computer science; MATLAB; Algorithm; Regression testing; Process (computing); Mutation; Software; Software development; Programming language","score_opus":0.0439389944210681,"score_gpt":0.3484687972418413,"score_spread":0.30452980282077324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964374341","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17479503,0.00079618586,0.8139892,0.00019307669,0.00014220517,0.00011530938,0.00034333585,0.0051957564,0.004429978],"genre_scores_gemma":[0.58562535,0.0001460835,0.4115942,0.000052262927,0.000022985569,0.000111838955,0.0005404163,0.00055646943,0.0013503747],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9908879,0.004055422,0.0006283991,0.0007611585,0.0033322044,0.00033483774],"domain_scores_gemma":[0.93242747,0.05085191,0.0027862915,0.007646435,0.0055410536,0.00074690743],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043593203,0.0012014547,0.00080651266,0.0024564457,0.00047541453,0.0012436694,0.003142649,0.0015938772,0.002513995],"category_scores_gemma":[0.054369103,0.0004640849,0.0006818953,0.0015274624,0.0012125896,0.0023418842,0.0011100158,0.0014141015,0.00056665635],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014897661,0.0008133163,0.040856384,0.00062302715,0.00025708857,0.00037459537,0.00035608897,0.30329874,0.047989376,0.039228242,0.0063817604,0.5583316],"study_design_scores_gemma":[0.000045196855,0.000435179,0.003945117,0.0000667914,0.000052049847,0.00026335966,0.000058663252,0.95158154,0.029340584,0.013053603,0.0011224907,0.000035374273],"about_ca_topic_score_codex":0.0020045282,"about_ca_topic_score_gemma":0.0019561583,"teacher_disagreement_score":0.0043593203,"about_ca_system_score_codex":0.00083270407,"about_ca_system_score_gemma":0.0010083392,"threshold_uncertainty_score":0.02305454},"labels":[],"label_agreement":null},{"id":"W2967858513","doi":"10.1145/3338906.3338908","title":"Concolic testing for models of state-based systems","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Concolic testing; Computer science; White-box testing; Model-based testing; Symbolic execution; Test case; Integration testing; System under test; Benchmark (surveying); Keyword-driven testing; Random testing; Unified Modeling Language; Manual testing; Test strategy; Code coverage; Non-regression testing; Unit testing; Context (archaeology); System testing; Regression testing; Software; Software engineering; Programming language; Software system; Machine learning; Software construction","score_opus":0.057612966011510496,"score_gpt":0.27455193970213226,"score_spread":0.21693897369062176,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2967858513","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029228136,0.00016460562,0.96477336,0.00019654848,0.00002278132,0.00011865962,0.0001832853,0.0023138332,0.0029988252],"genre_scores_gemma":[0.6836308,0.0004324426,0.30947143,0.00016726293,0.00004568517,0.0006542354,0.0009987308,0.0006279983,0.003971351],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968168,0.0013355956,0.00017189283,0.00028919592,0.001194605,0.00019189037],"domain_scores_gemma":[0.9911936,0.0063848887,0.00073474756,0.0010011261,0.0006053369,0.00008032244],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019498819,0.0010489463,0.0006182887,0.001213169,0.0004935142,0.0014490554,0.00137045,0.0011903549,0.0030683647],"category_scores_gemma":[0.012492581,0.00064265553,0.0011702051,0.00057376915,0.0016787171,0.0017389462,0.0012385,0.0015231551,0.00043972238],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017220044,0.0001316944,0.0017404528,0.00022319252,0.000056120898,0.0004075469,0.00032605993,0.7789432,0.009328918,0.1784128,0.0010439327,0.029213974],"study_design_scores_gemma":[0.000014940957,0.000032458433,0.00009204966,0.00002457643,0.000009043259,0.00005443764,0.000010016525,0.9676042,0.003438474,0.02712818,0.0015836314,0.000007936898],"about_ca_topic_score_codex":0.0060115084,"about_ca_topic_score_gemma":0.0068134796,"teacher_disagreement_score":0.0060115084,"about_ca_system_score_codex":0.0013980224,"about_ca_system_score_gemma":0.0017317145,"threshold_uncertainty_score":0.011953056},"labels":[],"label_agreement":null},{"id":"W2972465897","doi":"10.1007/s10009-019-00530-6","title":"Diversity of graph models and graph generators in mutation testing","year":2019,"lang":"en","type":"article","venue":"International Journal on Software Tools for Technology Transfer","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Emberi Eroforrások Minisztériuma; Natural Sciences and Engineering Research Council of Canada; Magyar Tudományos Akadémia; Budapesti Műszaki és Gazdaságtudományi Egyetem","keywords":"Computer science; Test suite; Predicate abstraction; Random testing; Graph; Model-based testing; Theoretical computer science; Predicate (mathematical logic); Software quality; Programming language; Test case; Software; Software engineering; Model checking; Machine learning; Software development","score_opus":0.03993430915760716,"score_gpt":0.26907315315342245,"score_spread":0.2291388439958153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2972465897","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6908224,0.00027521932,0.30586985,0.00030630786,0.000020099435,0.00013130173,0.000117149175,0.0005555672,0.001902111],"genre_scores_gemma":[0.95229214,0.00005983265,0.047064632,0.00003535986,0.00000872948,0.000056308225,0.00018852612,0.00010716499,0.00018725534],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.989608,0.006253354,0.00036268632,0.001012531,0.00237291,0.00039064343],"domain_scores_gemma":[0.9092877,0.07603114,0.0046633193,0.0061291573,0.0029042268,0.0009844657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00855693,0.000722795,0.0007449431,0.0033339872,0.0007154479,0.0017423342,0.0013956664,0.0016838738,0.00078528305],"category_scores_gemma":[0.053440645,0.0006858926,0.0009096464,0.0012620655,0.0027320383,0.0033950421,0.0023644597,0.00132902,0.000108006396],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051731896,0.0002756941,0.023433596,0.00014045092,0.000111942754,0.0004300984,0.0005780072,0.8733724,0.013035996,0.03121748,0.00045871997,0.0564284],"study_design_scores_gemma":[0.000036710004,0.00027554776,0.0018814297,0.000026727526,0.000032095097,0.0002607667,0.000103838465,0.9626543,0.0077962833,0.02644657,0.00046255146,0.000023138691],"about_ca_topic_score_codex":0.0013617001,"about_ca_topic_score_gemma":0.0015269404,"teacher_disagreement_score":0.00855693,"about_ca_system_score_codex":0.0013971373,"about_ca_system_score_gemma":0.0007958084,"threshold_uncertainty_score":0.045253932},"labels":[],"label_agreement":null},{"id":"W2977837980","doi":"10.1109/qrs.2019.00049","title":"Fault Detection in Timed FSM with Timeouts by SAT-Solving","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Scalability; Implementation; Lift (data mining); Finite-state machine; Fault injection; Fault detection and isolation; Constraint (computer-aided design); Model checking; Fault coverage; Distributed computing; Embedded system; Real-time computing; Programming language; Software; Data mining; Artificial intelligence; Actuator","score_opus":0.0050325935670965626,"score_gpt":0.20667492671879337,"score_spread":0.2016423331516968,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2977837980","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053854786,0.00009464635,0.94201344,0.0001466381,0.000027031505,0.00007930051,0.00020868567,0.0025189456,0.0010565186],"genre_scores_gemma":[0.4318866,0.0000982149,0.5658153,0.0001310111,0.000024475872,0.00021335934,0.00048276773,0.00024776364,0.0011004894],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99887985,0.0004722076,0.00008872781,0.0002351507,0.000219388,0.00010468309],"domain_scores_gemma":[0.9923375,0.006627435,0.00035329012,0.00036683673,0.0002470079,0.000067992645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012095678,0.000990816,0.00069455267,0.0007982608,0.00040945387,0.0009014828,0.0010699135,0.0009224038,0.0026397237],"category_scores_gemma":[0.006879531,0.0004956075,0.0013101827,0.00079392386,0.0011905296,0.0013913653,0.00075700076,0.0009515414,0.00027681483],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003171413,0.00014330768,0.002674936,0.00033663327,0.000121663325,0.00039309825,0.00017849614,0.8808647,0.015451687,0.025334705,0.0012421203,0.07294163],"study_design_scores_gemma":[0.000050816532,0.000038010094,0.00015566936,0.000013711038,0.000021178756,0.00005226777,0.000022652339,0.9753645,0.007997865,0.015691776,0.00058346795,0.00000812924],"about_ca_topic_score_codex":0.0042275167,"about_ca_topic_score_gemma":0.0059297173,"teacher_disagreement_score":0.0042275167,"about_ca_system_score_codex":0.0008000931,"about_ca_system_score_gemma":0.00125978,"threshold_uncertainty_score":0.008830726},"labels":[],"label_agreement":null},{"id":"W2978909621","doi":"10.1109/qrs.2019.00055","title":"Efficient Generation of Test Data with Extended Cardinality Constraints","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Cardinality (data modeling); Computer science; Theoretical computer science; Extension (predicate logic); Translation (biology); Graph; Algorithm; Programming language; Data mining","score_opus":0.05952479385163361,"score_gpt":0.28876779762455956,"score_spread":0.22924300377292595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2978909621","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.077968165,0.000077477765,0.9082741,0.0004134428,0.000077255245,0.0002099947,0.0014759483,0.009130171,0.0023733308],"genre_scores_gemma":[0.4391757,0.00007267765,0.5528129,0.00028395842,0.00004411596,0.00046331758,0.00402173,0.0013103484,0.0018152769],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971654,0.0006991004,0.00023397889,0.00038014379,0.0012950947,0.00022632683],"domain_scores_gemma":[0.98528916,0.010059474,0.000771554,0.0020886564,0.0015831048,0.00020804927],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001966521,0.0009527276,0.00091420807,0.001550105,0.00041804338,0.001285824,0.0020506177,0.000816661,0.0029871059],"category_scores_gemma":[0.016637962,0.00046568114,0.0010718488,0.0012998267,0.0011199093,0.002269853,0.0016891089,0.0012372921,0.00046935058],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012275368,0.0005673367,0.010201028,0.0008648558,0.00022792892,0.002430376,0.0008822661,0.27365416,0.08592177,0.11104742,0.017830709,0.49514467],"study_design_scores_gemma":[0.00021584677,0.0001457334,0.00062772684,0.00005303018,0.000056454108,0.0003975273,0.0000963589,0.8156962,0.09961324,0.07552885,0.0075154477,0.000053535532],"about_ca_topic_score_codex":0.0019017581,"about_ca_topic_score_gemma":0.0024056504,"teacher_disagreement_score":0.0029871059,"about_ca_system_score_codex":0.001144578,"about_ca_system_score_gemma":0.0017312749,"threshold_uncertainty_score":0.010400057},"labels":[],"label_agreement":null},{"id":"W2979344318","doi":"10.1007/978-3-030-31280-0_7","title":"Multiple Mutation Testing for Timed Finite State Machine with Timed Guards and Timeouts","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Finite-state machine; Process (computing); State (computer science); Test suite; Fault (geology); Mutation; Domain (mathematical analysis); Enumeration; Theoretical computer science; Fault injection; Model checking; Algorithm; Test case; Programming language; Software; Mathematics; Machine learning","score_opus":0.01965444124293541,"score_gpt":0.245311961391384,"score_spread":0.2256575201484486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2979344318","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06419433,0.0003120743,0.92888147,0.00021964959,0.0001717744,0.00006258258,0.00008203371,0.002526405,0.0035496654],"genre_scores_gemma":[0.7874867,0.00014612179,0.20838954,0.00010157213,0.000048460057,0.00007698168,0.0001374225,0.0003136831,0.0032995539],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.997456,0.0005424306,0.00014561016,0.00063153554,0.00093683583,0.0002875135],"domain_scores_gemma":[0.9926898,0.005623031,0.000406128,0.0005595914,0.0005207342,0.00020071019],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015298831,0.0010284152,0.0011251013,0.0010087572,0.00052311283,0.0010555243,0.001942707,0.0011265943,0.00227868],"category_scores_gemma":[0.006114334,0.00044065455,0.0013716565,0.0006928371,0.0019844822,0.0020577132,0.0013266078,0.0016140951,0.00026658026],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013600222,0.00039253396,0.0040248465,0.00072310393,0.00021697732,0.0036578104,0.00064244773,0.26050052,0.10349157,0.2934264,0.0043406948,0.32722306],"study_design_scores_gemma":[0.00010134463,0.00029486202,0.00043279625,0.000068816254,0.00009221313,0.00084423454,0.000057257148,0.7639062,0.051718734,0.17962949,0.0028045427,0.000049585018],"about_ca_topic_score_codex":0.0015111472,"about_ca_topic_score_gemma":0.0014684881,"teacher_disagreement_score":0.00227868,"about_ca_system_score_codex":0.0010972165,"about_ca_system_score_gemma":0.0012997251,"threshold_uncertainty_score":0.008090913},"labels":[],"label_agreement":null},{"id":"W2987704219","doi":"10.1145/3356773.3356810","title":"Workshop Summary","year":2019,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Software engineering; Software testing; Event (particle physics); Engineering management; Engineering; Library science; Software; Computer science; Programming language","score_opus":0.015181600319348384,"score_gpt":0.23504206514700132,"score_spread":0.21986046482765292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2987704219","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004391115,0.016643854,0.016791804,0.10669164,0.2875824,0.002885253,0.015193325,0.0034422418,0.54637843],"genre_scores_gemma":[0.020966377,0.009709966,0.008120783,0.022541225,0.032488305,0.0021509458,0.015658773,0.0016543133,0.8867092],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9966474,0.0007100393,0.00027111606,0.00056199083,0.0011859346,0.0006234097],"domain_scores_gemma":[0.989349,0.000738987,0.00032856961,0.00068111205,0.004541988,0.004360331],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0055374517,0.001314319,0.0007496779,0.001999672,0.002018821,0.006695829,0.0030944305,0.0030663102,0.3655499],"category_scores_gemma":[0.011181395,0.00042322703,0.0012579652,0.0014459677,0.00047467963,0.0031504387,0.0055884807,0.0034771417,0.23710988],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015934062,0.00006232304,0.00013212232,0.00043601016,0.000010093963,0.00011740819,0.0001391928,0.00014374114,0.0007347144,0.0023877537,0.9113961,0.08428122],"study_design_scores_gemma":[0.000020024223,0.000047028552,0.00025587072,0.00022354822,0.0000052475325,0.000059472175,0.0001573682,0.00004320446,0.00021920267,0.0009717774,0.99798834,0.000008789891],"about_ca_topic_score_codex":0.001558197,"about_ca_topic_score_gemma":0.0033374766,"teacher_disagreement_score":0.3655499,"about_ca_system_score_codex":0.0021728706,"about_ca_system_score_gemma":0.00602968,"threshold_uncertainty_score":0.90496606},"labels":[],"label_agreement":null},{"id":"W2987942974","doi":"10.48550/arxiv.1911.07444","title":"A Code Injection Method for Rapid Docker Image Building","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Code (set theory); Image (mathematics); Computer science; Operating system; Programming language; Artificial intelligence","score_opus":0.08449209376665238,"score_gpt":0.25005849032487165,"score_spread":0.16556639655821925,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2987942974","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005972979,0.000047519297,0.96201694,0.000093522205,0.00005190369,0.000106150444,0.000054322903,0.030284274,0.0013723047],"genre_scores_gemma":[0.13050735,0.00007957142,0.85703045,0.00026132626,0.0000402617,0.00031373746,0.00034659053,0.0067335577,0.004687154],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9979494,0.00032781373,0.0001791373,0.00041461593,0.0009159256,0.00021314509],"domain_scores_gemma":[0.9950275,0.0016068716,0.0004236699,0.0018856415,0.0008370971,0.000219222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014640049,0.000984899,0.00065062026,0.0012559359,0.00064282544,0.0010338323,0.002333558,0.0011945923,0.005775116],"category_scores_gemma":[0.0076743383,0.00089476566,0.0010016938,0.000637618,0.0014392773,0.0024385578,0.0025957946,0.0023548834,0.0027819816],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007898563,0.0006008875,0.0052401796,0.00035865922,0.00011403597,0.0008826555,0.0010987271,0.027479565,0.15054713,0.051020138,0.025167283,0.7367009],"study_design_scores_gemma":[0.0002071177,0.000384146,0.001731599,0.00010058969,0.00008405193,0.0013507607,0.00010549403,0.64903307,0.27318117,0.028498564,0.04514071,0.00018272095],"about_ca_topic_score_codex":0.0010263377,"about_ca_topic_score_gemma":0.0011279505,"teacher_disagreement_score":0.005775116,"about_ca_system_score_codex":0.0006697362,"about_ca_system_score_gemma":0.0012147843,"threshold_uncertainty_score":0.019319713},"labels":[],"label_agreement":null},{"id":"W2989862949","doi":"10.1109/models.2019.00-12","title":"Towards System-Level Testing with Coverage Guarantees for Autonomous Vehicles","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Test suite; Domain (mathematical analysis); Safety assurance; Geospatial analysis; Component (thermodynamics); Suite; Abstraction; Graph; Context (archaeology); Field (mathematics); Distributed computing; Test case; Software engineering; Reliability engineering; Theoretical computer science; Machine learning; Engineering","score_opus":0.036738859316262244,"score_gpt":0.24882340102546516,"score_spread":0.2120845417092029,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2989862949","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07860859,0.00016029834,0.91848963,0.00019739612,0.00001161574,0.000057387333,0.00012729,0.0016064183,0.0007413687],"genre_scores_gemma":[0.79839337,0.00015937186,0.20025927,0.00009464766,0.000023307091,0.00013176756,0.00038163931,0.00029746542,0.00025918314],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99172276,0.0030887914,0.00044665602,0.00088561117,0.0033153286,0.0005408988],"domain_scores_gemma":[0.9739057,0.01650721,0.0028885177,0.0044431123,0.0018562104,0.00039923593],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040823156,0.0010066441,0.00071798975,0.0020830154,0.00035896632,0.0020127008,0.001709643,0.0012907628,0.00092109526],"category_scores_gemma":[0.028691048,0.00068295473,0.0012043994,0.00093574735,0.0026251725,0.002550546,0.0023684022,0.0016545667,0.00017296529],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025812778,0.00022184622,0.012324033,0.00036229842,0.00015862439,0.00033937534,0.0006127614,0.8236723,0.044770047,0.05608178,0.0007474145,0.06045144],"study_design_scores_gemma":[0.000017325752,0.00013183372,0.0010763521,0.00003603436,0.000031485946,0.00008484958,0.000057386584,0.9462786,0.013721729,0.03783854,0.00071041717,0.000015481697],"about_ca_topic_score_codex":0.0029098284,"about_ca_topic_score_gemma":0.0023932343,"teacher_disagreement_score":0.0040823156,"about_ca_system_score_codex":0.0011425627,"about_ca_system_score_gemma":0.0014902616,"threshold_uncertainty_score":0.021589637},"labels":[],"label_agreement":null},{"id":"W2990853981","doi":"10.1109/models-c.2019.00008","title":"Querying Automotive System Models and Safety Artifacts with MMINT and Viatra","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Traceability; Computer science; Automotive industry; Domain (mathematical analysis); Software engineering; Hazard and operability study; Systems engineering; Engineering","score_opus":0.010829436579208016,"score_gpt":0.2034567904169108,"score_spread":0.1926273538377028,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2990853981","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03435163,0.000254436,0.92742735,0.00060194876,0.000029262803,0.00028834344,0.0013036212,0.03063903,0.005104414],"genre_scores_gemma":[0.4485512,0.0005147246,0.53634495,0.0005009308,0.000053899475,0.00043420054,0.005148258,0.0047372235,0.003714688],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98979867,0.0030754718,0.0015745694,0.0010812965,0.0040638275,0.00040616756],"domain_scores_gemma":[0.98490983,0.00565276,0.0013141864,0.0065131127,0.001399239,0.00021080852],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008257364,0.0016837767,0.0011610205,0.0029641027,0.0008753067,0.0057428475,0.0028498527,0.001870076,0.0037377062],"category_scores_gemma":[0.02850467,0.0012643309,0.0020334646,0.0017899309,0.0015707607,0.009427326,0.006985236,0.0012557147,0.000990783],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015637723,0.000711742,0.043767292,0.0024727175,0.0007770784,0.0037913173,0.010664575,0.09286714,0.06817794,0.30232188,0.026540121,0.44634444],"study_design_scores_gemma":[0.0002095089,0.00056151766,0.0056427317,0.000511773,0.00037590598,0.0033195028,0.00248168,0.6334824,0.099925354,0.09590853,0.15726696,0.00031420102],"about_ca_topic_score_codex":0.0057374043,"about_ca_topic_score_gemma":0.008867829,"teacher_disagreement_score":0.008257364,"about_ca_system_score_codex":0.0015065244,"about_ca_system_score_gemma":0.0014792876,"threshold_uncertainty_score":0.04366964},"labels":[],"label_agreement":null},{"id":"W2994711059","doi":"10.1002/stvr.1721","title":"Leveraging metamorphic testing to automatically detect inconsistencies in code generator families","year":2019,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Agence Nationale de la Recherche","keywords":"Computer science; Software quality; Oracle; Software; Leverage (statistics); Regression testing; Code (set theory); Code coverage; Unreachable code; Software development; Code generation; Programming language; Redundant code; Software construction; Set (abstract data type); Operating system; Artificial intelligence","score_opus":0.03641849049665075,"score_gpt":0.2574023251371866,"score_spread":0.22098383464053584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2994711059","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4932825,0.00028323138,0.49618748,0.00016575487,0.000024408562,0.00015069601,0.0002188367,0.008405702,0.0012813088],"genre_scores_gemma":[0.87813514,0.00005324172,0.120824754,0.000057921567,0.000012926441,0.00006927011,0.00038745799,0.00020660601,0.00025281784],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964659,0.0009975854,0.00031637365,0.0008288821,0.0012069517,0.00018424704],"domain_scores_gemma":[0.976344,0.012546669,0.005570724,0.0024820275,0.0026740828,0.0003825188],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002627858,0.00053986127,0.0006672531,0.0037166926,0.00030464842,0.00086493,0.0011994895,0.00063852046,0.00053502194],"category_scores_gemma":[0.017690634,0.0003904501,0.0005939456,0.0010644413,0.0007113995,0.0008786203,0.0011354052,0.00067203795,0.00020296278],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007035004,0.0004319251,0.21237373,0.0003667823,0.0002590825,0.001517168,0.00077849126,0.17012355,0.11569347,0.0084612295,0.0022529871,0.4870381],"study_design_scores_gemma":[0.000029660749,0.00022270545,0.016867412,0.000045015655,0.000044968758,0.0006775658,0.000054094828,0.9500443,0.026262224,0.004777249,0.00094486127,0.000029917108],"about_ca_topic_score_codex":0.0011428174,"about_ca_topic_score_gemma":0.0012762825,"teacher_disagreement_score":0.0037166926,"about_ca_system_score_codex":0.00045127323,"about_ca_system_score_gemma":0.00065874326,"threshold_uncertainty_score":0.013897657},"labels":[],"label_agreement":null},{"id":"W3000541349","doi":"10.1109/ase.2019.00132","title":"mCUTE: A Model-Level Concolic Unit Testing Engine for UML State Machines","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Concolic testing; Computer science; Unified Modeling Language; Model-based testing; Applications of UML; Finite-state machine; Programming language; Embedded system; Software; Symbolic execution; Test case","score_opus":0.09458609736607745,"score_gpt":0.30561833662574867,"score_spread":0.2110322392596712,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3000541349","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0069455877,0.00011266851,0.9106756,0.00008089509,0.00003044928,0.00018417003,0.0006038466,0.07919531,0.0021715635],"genre_scores_gemma":[0.2113703,0.00036310256,0.7627823,0.00025719678,0.000048828388,0.00075323967,0.0042113685,0.01510703,0.0051066387],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99860567,0.0003497993,0.00010733905,0.00019175847,0.0006274269,0.00011798877],"domain_scores_gemma":[0.9967686,0.0021201952,0.0002994059,0.00044586597,0.0003066049,0.000059371512],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016770741,0.0015437969,0.0007672902,0.0019828405,0.00046168402,0.0013280198,0.0023412893,0.0016180583,0.0077016433],"category_scores_gemma":[0.009586788,0.00095812476,0.0014894153,0.0006725259,0.000961615,0.0019737498,0.0015907397,0.0013685686,0.0023990397],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010025089,0.0006991092,0.013069709,0.002389212,0.0004664766,0.0016356678,0.0011297904,0.25996232,0.06719417,0.091726236,0.0517133,0.5090115],"study_design_scores_gemma":[0.00012950471,0.00015318024,0.0010573934,0.00016232317,0.000064870124,0.00054947136,0.00003715478,0.902125,0.047909252,0.01656735,0.031172125,0.00007234977],"about_ca_topic_score_codex":0.0029363965,"about_ca_topic_score_gemma":0.0034700246,"teacher_disagreement_score":0.0077016433,"about_ca_system_score_codex":0.00075427844,"about_ca_system_score_gemma":0.0015064634,"threshold_uncertainty_score":0.025764525},"labels":[],"label_agreement":null},{"id":"W3004099993","doi":"10.1145/2499370.2462168","title":"Dynamic determinacy analysis","year":2013,"lang":"en","type":"article","venue":"ACM SIGPLAN Notices","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Static analysis; Programming language; JavaScript; Soundness; Determinacy; Program analysis; Functional programming; Leverage (statistics); Source code; Scalability; Theoretical computer science; Operating system; Artificial intelligence","score_opus":0.014623327021583827,"score_gpt":0.2669698366788959,"score_spread":0.2523465096573121,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004099993","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012367833,0.00026752698,0.96988297,0.00049735507,0.00011475497,0.000142611,0.00042234903,0.0042394465,0.012065201],"genre_scores_gemma":[0.49273276,0.0007673462,0.48349905,0.0009777179,0.00035191272,0.00066983077,0.0013292122,0.0033604975,0.016311666],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99194735,0.0011429307,0.0004072747,0.0016002789,0.003644374,0.0012578259],"domain_scores_gemma":[0.9877309,0.005232029,0.0009962787,0.0032276104,0.0025814509,0.00023177134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032345932,0.0012983134,0.0012372336,0.004511478,0.0028730102,0.0033508488,0.0025724862,0.0013922262,0.0075967438],"category_scores_gemma":[0.016101703,0.001268556,0.002555093,0.0022898668,0.0042172195,0.0063414993,0.0058458527,0.0037582442,0.0024061745],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048641875,0.00020233737,0.009541548,0.000569882,0.0001597491,0.001148441,0.0015393323,0.035875645,0.02794195,0.6719609,0.010884722,0.23968905],"study_design_scores_gemma":[0.000050679617,0.000094928815,0.0015993806,0.00023562287,0.00023762426,0.0008895024,0.0002588385,0.14744346,0.056805916,0.73044944,0.061780196,0.00015443937],"about_ca_topic_score_codex":0.003787965,"about_ca_topic_score_gemma":0.0032487586,"teacher_disagreement_score":0.0075967438,"about_ca_system_score_codex":0.0021575799,"about_ca_system_score_gemma":0.003974565,"threshold_uncertainty_score":0.025413573},"labels":[],"label_agreement":null},{"id":"W3004493192","doi":"10.1016/j.jss.2020.110542","title":"On testing machine learning programs","year":2020,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":198,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Transformative learning; Domain (mathematical analysis); Software testing; Software engineering; Test strategy; Field (mathematics); Software; Witness; Data science; Artificial intelligence; Machine learning; Psychology","score_opus":0.027723359479573075,"score_gpt":0.23802568780602595,"score_spread":0.21030232832645288,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004493192","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.072855555,0.0022477908,0.8975637,0.0029617648,0.0003947412,0.00017683693,0.00038211167,0.0067100557,0.016707418],"genre_scores_gemma":[0.60739297,0.0010105851,0.37447846,0.0013349588,0.00048761402,0.00025217267,0.0019440028,0.0018381759,0.011261054],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9829312,0.007889721,0.00096148695,0.0022697367,0.004868865,0.0010789941],"domain_scores_gemma":[0.8986333,0.081924744,0.0015943273,0.012482975,0.0044952403,0.0008694676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054384726,0.0024126077,0.0016896568,0.003910051,0.0011441768,0.001782613,0.0046155266,0.002585734,0.008551338],"category_scores_gemma":[0.06106046,0.0012788245,0.0020479301,0.0034760348,0.003617962,0.0085479105,0.004396002,0.0033997195,0.0010704212],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010337692,0.0006561562,0.010375278,0.0006831214,0.00030510838,0.00068269187,0.00049676246,0.12923576,0.0093428325,0.121542014,0.021429867,0.70421654],"study_design_scores_gemma":[0.00012662157,0.0002885397,0.0025899312,0.00023255014,0.00017368874,0.00043887526,0.0001534572,0.6812806,0.012298166,0.2944688,0.007908997,0.000039790386],"about_ca_topic_score_codex":0.008490797,"about_ca_topic_score_gemma":0.012065079,"teacher_disagreement_score":0.008551338,"about_ca_system_score_codex":0.0021873366,"about_ca_system_score_gemma":0.001854638,"threshold_uncertainty_score":0.028761744},"labels":[],"label_agreement":null},{"id":"W3006547141","doi":"10.1109/issre.2019.00018","title":"Engineering a Better Fuzzer with Synergically Integrated Optimizations","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Fuzz testing; Computer science; Block (permutation group theory); Granularity; Scheduling (production processes); Knapsack problem; Parallel computing; Algorithm; Mathematical optimization; Operating system; Software","score_opus":0.0037741349157259852,"score_gpt":0.17514663970216207,"score_spread":0.1713725047864361,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3006547141","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23261929,0.00084019493,0.7394863,0.00054922793,0.00012752827,0.00042629262,0.00020240262,0.02157363,0.004175175],"genre_scores_gemma":[0.5605812,0.00023001934,0.43529257,0.00034617155,0.000036211684,0.00016269149,0.00033988562,0.0010902375,0.0019209629],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978054,0.0003402475,0.00021244184,0.00041948457,0.0008983408,0.00032414528],"domain_scores_gemma":[0.99685293,0.00099433,0.00030356355,0.001265571,0.0004752214,0.00010831554],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024601114,0.0015451486,0.0007800422,0.0014102289,0.00044523898,0.0013200471,0.0015639821,0.000996896,0.0017173543],"category_scores_gemma":[0.005683573,0.00070986827,0.0011744709,0.00055559643,0.001027107,0.002609718,0.001994802,0.00158546,0.00046827938],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010282166,0.0010485998,0.028787198,0.0008281184,0.0007661692,0.00086840295,0.0006478548,0.21152835,0.24886405,0.020576283,0.006112277,0.47894448],"study_design_scores_gemma":[0.00017075361,0.00090934895,0.0066439332,0.00013718882,0.0006060055,0.0006673958,0.00019216315,0.8177324,0.14359811,0.014897296,0.014301068,0.00014429902],"about_ca_topic_score_codex":0.002207561,"about_ca_topic_score_gemma":0.003349024,"teacher_disagreement_score":0.0024601114,"about_ca_system_score_codex":0.0007664728,"about_ca_system_score_gemma":0.0017121768,"threshold_uncertainty_score":0.013010442},"labels":[],"label_agreement":null},{"id":"W3011348730","doi":"10.1109/ibf50092.2020.9034821","title":"Exploring the Differences between Plausible and Correct Patches at Fine-Grained Level","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Correctness; Abstraction; Invariant (physics); Test case; Programming language; Artificial intelligence; Machine learning; Mathematics","score_opus":0.3754039426601999,"score_gpt":0.2731418934620426,"score_spread":0.10226204919815729,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3011348730","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7597989,0.00035043893,0.23340778,0.00018451536,0.00002213949,0.00017122574,0.00040095727,0.002715495,0.0029487144],"genre_scores_gemma":[0.93585414,0.000079893616,0.06264446,0.000042502787,0.000004734811,0.000056566623,0.00037333512,0.0005773139,0.00036716383],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99613,0.00090760214,0.00023957156,0.00070757035,0.0017012755,0.00031403304],"domain_scores_gemma":[0.95799255,0.028636888,0.0032015317,0.007537012,0.0023271812,0.00030478518],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003770155,0.0005557916,0.0004629545,0.0010380455,0.0003312761,0.0013760377,0.0011913844,0.0008274643,0.0018240041],"category_scores_gemma":[0.03740727,0.00051845395,0.0005507817,0.00053407275,0.0014471738,0.0022239042,0.000944728,0.0012062932,0.0003291483],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016150355,0.0006220036,0.2256907,0.0020452004,0.00041097068,0.0030731857,0.0066799438,0.17488414,0.21580836,0.026512599,0.0027827364,0.33987513],"study_design_scores_gemma":[0.00016161715,0.0016079673,0.13448252,0.0003496601,0.0004791989,0.0029933113,0.0024255558,0.6444881,0.1504511,0.04564225,0.016755939,0.00016276547],"about_ca_topic_score_codex":0.0010122851,"about_ca_topic_score_gemma":0.0019279445,"teacher_disagreement_score":0.003770155,"about_ca_system_score_codex":0.0005300993,"about_ca_system_score_gemma":0.00056110666,"threshold_uncertainty_score":0.019938767},"labels":[],"label_agreement":null},{"id":"W3013054130","doi":"10.1002/smr.415","title":"Regression test suite reduction based on SDL models of system requirements","year":2009,"lang":"en","type":"article","venue":"Journal of Software Maintenance and Evolution Research and Practice","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Test suite; Regression testing; Computer science; Suite; Regression analysis; Reduction (mathematics); Test (biology); Set (abstract data type); Test case; Regression; Programming language; Data mining; Statistics; Machine learning; Mathematics; Software; Software development","score_opus":0.08214894951886607,"score_gpt":0.37058079961742646,"score_spread":0.28843185009856037,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3013054130","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08862972,0.00010270334,0.90460455,0.00025107083,0.000018740271,0.0002337389,0.00018191023,0.0042423983,0.0017351442],"genre_scores_gemma":[0.57688355,0.00009325599,0.41997018,0.000111386646,0.000019719293,0.00036018784,0.0010503975,0.0005156959,0.0009956583],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99411976,0.0027947763,0.00022399369,0.00045910777,0.0021188103,0.0002836443],"domain_scores_gemma":[0.98318696,0.011554397,0.0011829128,0.0022738485,0.0016229785,0.00017896772],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024525765,0.001087188,0.0010089716,0.002159143,0.00030316785,0.00092389283,0.001366118,0.0005345387,0.0017517088],"category_scores_gemma":[0.022115652,0.0007131732,0.001564102,0.00082097587,0.00067031663,0.0012256001,0.0012646586,0.0014527715,0.00035950416],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044861395,0.0005722817,0.007399554,0.00020867301,0.00018407161,0.00050526636,0.0002526938,0.7054731,0.029416611,0.012583502,0.002336735,0.24061893],"study_design_scores_gemma":[0.000019997962,0.00006559143,0.00038688612,0.000009914882,0.000018082545,0.00006888091,0.000013314274,0.9898413,0.0056624566,0.0032980845,0.000606021,0.000009437139],"about_ca_topic_score_codex":0.0032983597,"about_ca_topic_score_gemma":0.0031039817,"teacher_disagreement_score":0.0032983597,"about_ca_system_score_codex":0.0011272588,"about_ca_system_score_gemma":0.0013905539,"threshold_uncertainty_score":0.012970626},"labels":[],"label_agreement":null},{"id":"W3013655954","doi":"10.1002/stvr.380","title":"Automated discovery of state transitions and their functions in source code","year":2007,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Engineering and Physical Sciences Research Council; National Aeronautics and Space Administration","keywords":"Computer science; State (computer science); Source code; Finite-state machine; Reverse engineering; Set (abstract data type); Code (set theory); Programming language; Software; Software engineering; Transition (genetics); Theoretical computer science","score_opus":0.01937952920259313,"score_gpt":0.2572724153819903,"score_spread":0.2378928861793972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3013655954","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23582442,0.00018941658,0.7460938,0.00017431524,0.00002270662,0.00010429418,0.00018778002,0.015963182,0.001440115],"genre_scores_gemma":[0.75985944,0.000121943776,0.23805591,0.000027250184,0.000006801225,0.00007062204,0.00044467676,0.00069743075,0.0007159485],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983682,0.0005929438,0.00010055019,0.00022591215,0.00060606166,0.00010638159],"domain_scores_gemma":[0.98507,0.010435237,0.0017341479,0.0017193288,0.00094260246,0.00009865344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015079489,0.0006328678,0.00045221113,0.0014044484,0.00040591977,0.0010437056,0.0010110466,0.0008669998,0.0013026537],"category_scores_gemma":[0.013248674,0.0005962955,0.00055026927,0.0005948549,0.001234705,0.0012705142,0.00080484786,0.000954599,0.0004285913],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008378351,0.0003854986,0.030090868,0.0008114832,0.00014064476,0.0023185979,0.0020776694,0.20615298,0.18958293,0.059655555,0.0024098088,0.5055362],"study_design_scores_gemma":[0.00003278826,0.000093572766,0.002062293,0.00005742852,0.000041078456,0.00043209133,0.000058243168,0.86467665,0.11316794,0.016954139,0.0023907886,0.000032923934],"about_ca_topic_score_codex":0.00188932,"about_ca_topic_score_gemma":0.0015839919,"teacher_disagreement_score":0.00188932,"about_ca_system_score_codex":0.0005454903,"about_ca_system_score_gemma":0.0010621945,"threshold_uncertainty_score":0.007974923},"labels":[],"label_agreement":null},{"id":"W3021368500","doi":"10.1145/2858965.2814297","title":"SATCheck: SAT-directed stateless model checking for SC and TSO","year":2015,"lang":"en","type":"article","venue":"ACM SIGPLAN Notices","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Science Foundation","keywords":"Computer science; Model checking; Scalability; Programming language; Stateless protocol; Parallel computing; Memory model; Thread (computing); Concurrency; Shared memory; Operating system","score_opus":0.10695654727802961,"score_gpt":0.3183851898259565,"score_spread":0.21142864254792687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3021368500","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020359462,0.00013186,0.94729435,0.00031310463,0.00007017887,0.00016994329,0.0010381527,0.028044142,0.0025789153],"genre_scores_gemma":[0.3770478,0.00014768488,0.6137108,0.00043413544,0.00005004663,0.00044868916,0.0027116728,0.0024968968,0.0029522337],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99593383,0.001426052,0.0003134856,0.0007820696,0.0012012857,0.00034334473],"domain_scores_gemma":[0.98551744,0.009908098,0.0008752775,0.0023125515,0.0012194567,0.00016716063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029032973,0.0015632102,0.000879745,0.0014788638,0.0009520021,0.0017281603,0.0031906518,0.0011491538,0.0073955194],"category_scores_gemma":[0.015314298,0.0011146729,0.0030867588,0.0011696556,0.0027231716,0.0034375926,0.002389651,0.0023593418,0.0010914328],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014005599,0.00042404898,0.011470334,0.001443798,0.0004883283,0.0006282809,0.0005549598,0.6265643,0.023250813,0.14209238,0.01912418,0.17255801],"study_design_scores_gemma":[0.00016811064,0.00011287728,0.0002729577,0.0000553604,0.00006893177,0.000079141304,0.00005149479,0.9357536,0.016310666,0.04343346,0.0036671076,0.00002625255],"about_ca_topic_score_codex":0.014357806,"about_ca_topic_score_gemma":0.03031984,"teacher_disagreement_score":0.014357806,"about_ca_system_score_codex":0.0019802765,"about_ca_system_score_gemma":0.005421554,"threshold_uncertainty_score":0.02854848},"labels":[],"label_agreement":null},{"id":"W3025363231","doi":"10.1039/d0sc01232g","title":"Demonstration of the utility of DOS-derived fragment libraries for rapid hit derivatisation in a multidirectional fashion","year":2020,"lang":"en","type":"article","venue":"Chemical Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Structural Genomics Consortium","funders":"Biotechnology and Biological Sciences Research Council; Novartis Pharma; Medical Research Council; Ministero dello Sviluppo Economico; Innovative Medicines Initiative; Royal Society; Ontario Ministry of Economic Development and Innovation; Diamond Light Source; Wellcome Trust; Fundação de Amparo à Pesquisa do Estado de São Paulo; Genome Canada; Department of Biochemistry, University of Cambridge; AbbVie; European Federation of Pharmaceutical Industries and Associations; Merck KGaA; Meso Scale Diagnostics; Engineering and Physical Sciences Research Council; AstraZeneca; Takeda Pharmaceuticals U.S.A.; University of Warwick; Pfizer","keywords":"Fragment (logic); Computer science; Chemistry; Programming language","score_opus":0.038923748058163746,"score_gpt":0.25441191038409106,"score_spread":0.21548816232592732,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3025363231","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.816927,0.0015452723,0.16435829,0.00034223046,0.000045818015,0.00044584344,0.0015226868,0.002079654,0.012733206],"genre_scores_gemma":[0.9049453,0.0012135742,0.087468766,0.00010452454,0.000012754514,0.0001891926,0.001864458,0.00015416069,0.004047227],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999706,0.000046427984,0.000021051033,0.000049292183,0.00013172453,0.00004545159],"domain_scores_gemma":[0.9996724,0.00011075434,0.000063878666,0.00007471692,0.000043706277,0.00003454182],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00054169865,0.00045221928,0.00036548253,0.00050758215,0.00021373935,0.00053606887,0.00042087503,0.00024385029,0.0021600176],"category_scores_gemma":[0.00094407605,0.00021310995,0.00027447692,0.0005672076,0.00034687392,0.00044643835,0.00073878607,0.00077183766,0.00067261653],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002284627,0.000085598054,0.000625005,0.00010252271,0.000018734252,0.00018246345,0.00010196954,0.0039393413,0.9450531,0.0023607584,0.00029675124,0.047005344],"study_design_scores_gemma":[0.000039362727,0.0010432663,0.0007294652,0.000010303224,0.000018948234,0.0004501214,0.00003252662,0.0041063135,0.986704,0.0005607549,0.0062887194,0.000016106822],"about_ca_topic_score_codex":0.00037555996,"about_ca_topic_score_gemma":0.0009037477,"teacher_disagreement_score":0.0021600176,"about_ca_system_score_codex":0.00032925562,"about_ca_system_score_gemma":0.00038981432,"threshold_uncertainty_score":0.0072259903},"labels":[],"label_agreement":null},{"id":"W3029402327","doi":"10.1145/2578855.2535857","title":"Symbolic optimization with SMT solvers","year":2014,"lang":"en","type":"article","venue":"ACM SIGPLAN Notices","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Satisfiability modulo theories; Symbolic execution; Programming language; Satisfiability; Set (abstract data type); Function (biology); Software; Theoretical computer science; Algorithm","score_opus":0.012830927648873825,"score_gpt":0.2239901630927133,"score_spread":0.21115923544383947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029402327","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040891673,0.00034378213,0.9837753,0.00024297684,0.000057926994,0.000065406304,0.00014732483,0.002390088,0.0088880025],"genre_scores_gemma":[0.14059074,0.0006603634,0.8514326,0.00025785388,0.000105423824,0.00048264267,0.00061722816,0.001180581,0.0046725934],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99811506,0.00069274317,0.00012031689,0.00023102309,0.0006772652,0.00016365421],"domain_scores_gemma":[0.9969875,0.0022446918,0.00019205415,0.00027900282,0.00024402732,0.00005279047],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014737339,0.0018036042,0.0011890772,0.001415159,0.0005261568,0.0016951794,0.0013596015,0.0012521244,0.008946152],"category_scores_gemma":[0.0064303875,0.0007298065,0.0019596666,0.0016973458,0.0014591529,0.0013725837,0.0020018353,0.0023242615,0.00226223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010092181,0.000065582826,0.0004646434,0.00041319575,0.00007948019,0.00009574806,0.00006938921,0.81204873,0.0027248461,0.085959576,0.0031456153,0.09483229],"study_design_scores_gemma":[0.000032312815,0.000019901765,0.000039321745,0.000028257911,0.0000138489195,0.000018370578,0.000011928636,0.9577589,0.0013412826,0.037077986,0.0036516215,0.000006190738],"about_ca_topic_score_codex":0.0032419933,"about_ca_topic_score_gemma":0.0048244167,"teacher_disagreement_score":0.008946152,"about_ca_system_score_codex":0.0015366772,"about_ca_system_score_gemma":0.0023083768,"threshold_uncertainty_score":0.02992785},"labels":[],"label_agreement":null},{"id":"W3043367330","doi":"10.1145/3395363.3397386","title":"Automated repair of feature interaction failures in automated driving systems","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"H2020 European Research Council; Natural Sciences and Engineering Research Council of Canada; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Computer science; Feature (linguistics); Reliability engineering; Distributed computing; Embedded system; Artificial intelligence; Engineering","score_opus":0.020375712857958212,"score_gpt":0.2789408929585433,"score_spread":0.2585651801005851,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3043367330","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5452478,0.00061694125,0.43930826,0.00016847267,0.000037300822,0.00018429526,0.00012608024,0.013088893,0.0012219773],"genre_scores_gemma":[0.91917115,0.00007371391,0.079823494,0.000040620656,0.000008132069,0.00003331462,0.00012343415,0.00015726249,0.00056874275],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9985599,0.0003344834,0.000100438185,0.0002639318,0.0005928649,0.00014846348],"domain_scores_gemma":[0.9950547,0.0020026595,0.0009749559,0.001110782,0.0007149213,0.00014190836],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009224838,0.00075009844,0.00063249335,0.00097173627,0.00054266525,0.00040909275,0.0015629415,0.000806411,0.0009759575],"category_scores_gemma":[0.0048660296,0.00034521092,0.00042498042,0.00047300904,0.00074227905,0.00087869825,0.0007661895,0.00058996433,0.0002524237],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004909135,0.00041360772,0.01975059,0.00045534808,0.00009804964,0.0011121712,0.0010132134,0.30499798,0.14161675,0.0021121872,0.0025864034,0.52535284],"study_design_scores_gemma":[0.00007657699,0.0006306237,0.008446323,0.000034699784,0.00006668045,0.0009019636,0.00029095646,0.8942962,0.08696764,0.0031669254,0.0050658053,0.000055635388],"about_ca_topic_score_codex":0.0035086514,"about_ca_topic_score_gemma":0.0032621147,"teacher_disagreement_score":0.0035086514,"about_ca_system_score_codex":0.00049797737,"about_ca_system_score_gemma":0.00079431816,"threshold_uncertainty_score":0.0069764853},"labels":[],"label_agreement":null},{"id":"W3043761819","doi":"10.1145/3395363.3397369","title":"CoCoNuT: combining context-aware neural translation models using ensemble for program repair","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":329,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Facebook","keywords":"Computer science; Context (archaeology); Machine translation; Artificial intelligence; Programming language; Machine learning; Software engineering; Natural language processing","score_opus":0.1833401154001179,"score_gpt":0.3325084569739183,"score_spread":0.14916834157380038,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3043761819","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07254525,0.0078947125,0.8362793,0.0011403757,0.0016329691,0.0003880895,0.0035206347,0.06595095,0.010647712],"genre_scores_gemma":[0.529006,0.0024560157,0.4258242,0.0016033769,0.00066902686,0.0006070838,0.01775899,0.0042372,0.017838111],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99895227,0.00024103932,0.00005854419,0.00044036863,0.00019761566,0.00011011048],"domain_scores_gemma":[0.9983,0.0007421717,0.000077209166,0.0002646797,0.0005346805,0.00008126622],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014032264,0.0024294064,0.0018665964,0.002025556,0.0010502128,0.0017187275,0.0025508336,0.002208497,0.0061764247],"category_scores_gemma":[0.004590666,0.0005311625,0.0012790821,0.00216328,0.000472692,0.0027305812,0.0019859371,0.002282435,0.0038244564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007443565,0.0004676324,0.0025532704,0.0004507138,0.000490029,0.0004950628,0.00018985406,0.12187217,0.011676097,0.0024060647,0.035318077,0.8233366],"study_design_scores_gemma":[0.000036148314,0.00010327122,0.00036496442,0.000033582393,0.00010204997,0.000093709474,0.0000538503,0.986635,0.0038851448,0.0041360008,0.0045252023,0.000030999745],"about_ca_topic_score_codex":0.01933593,"about_ca_topic_score_gemma":0.03235472,"teacher_disagreement_score":0.01933593,"about_ca_system_score_codex":0.0009610077,"about_ca_system_score_gemma":0.0017729796,"threshold_uncertainty_score":0.038446784},"labels":[],"label_agreement":null},{"id":"W306569659","doi":"10.1016/b978-0-12-800160-8.00002-4","title":"Automated Extraction of GUI Models for Testing","year":2014,"lang":"en","type":"book-chapter","venue":"Advances in computers","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Graphical user interface; Software engineering; Process (computing); Graphical user interface testing; Domain (mathematical analysis); Model-based testing; Process modeling; Software; Data mining; User interface; Test case; Machine learning; Programming language; Work in process; User interface design; Engineering","score_opus":0.044030852554028777,"score_gpt":0.3119502890067512,"score_spread":0.2679194364527224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W306569659","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007863296,0.0010741543,0.9588418,0.00025286648,0.000065848326,0.00017388261,0.0020197446,0.022626936,0.007081449],"genre_scores_gemma":[0.13022107,0.0014067271,0.8459757,0.00014194485,0.000049780025,0.0002559734,0.010792954,0.0044108545,0.0067450237],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990004,0.00021572683,0.00006939963,0.00015372201,0.00049296045,0.00006781914],"domain_scores_gemma":[0.99706036,0.0016414087,0.00016229818,0.0006880614,0.00040982242,0.000038057744],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00075586466,0.0019219302,0.001099718,0.0026023064,0.00041321208,0.0021825326,0.002204922,0.0010475856,0.009617352],"category_scores_gemma":[0.005772149,0.0011345807,0.002028445,0.001981747,0.0005814149,0.0027299451,0.0015472347,0.0018462843,0.0049065487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022469246,0.00020821262,0.0023619933,0.0012856934,0.00012830984,0.00079673954,0.00031098802,0.06923772,0.03803435,0.0539217,0.03502637,0.79846334],"study_design_scores_gemma":[0.000058950856,0.00009631181,0.0015138087,0.00039833278,0.00014691861,0.00097086496,0.00011168008,0.7950172,0.059149433,0.07993292,0.06253746,0.00006617047],"about_ca_topic_score_codex":0.0027604694,"about_ca_topic_score_gemma":0.004848186,"teacher_disagreement_score":0.009617352,"about_ca_system_score_codex":0.00084648834,"about_ca_system_score_gemma":0.0010106445,"threshold_uncertainty_score":0.032173276},"labels":[],"label_agreement":null},{"id":"W3081900038","doi":"","title":"Unit Test Effort Prioritization Using Combined Datasets and Deep Learning: A Cross-Systems Validation.","year":2020,"lang":"en","type":"article","venue":"Software Engineering and Knowledge Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec","funders":"","keywords":"Prioritization; Computer science; Test (biology); Artificial intelligence; Unit (ring theory); Unit testing; Cross-validation; Machine learning; Data mining; Engineering; Software; Mathematics","score_opus":0.02080212164416674,"score_gpt":0.25420117283484395,"score_spread":0.23339905119067722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3081900038","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94164914,0.0011776411,0.042875588,0.00017663497,0.00021457915,0.00035651296,0.009109995,0.0028785248,0.0015612798],"genre_scores_gemma":[0.9374619,0.000112699316,0.041986253,0.00009896321,0.00003551196,0.0003376145,0.018678673,0.00034996544,0.0009382972],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99312663,0.0034251036,0.00054835953,0.0017957859,0.0008539257,0.0002500909],"domain_scores_gemma":[0.9573489,0.023803202,0.0024395788,0.00836331,0.007361762,0.00068335485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012582537,0.0015459579,0.0008807543,0.0026473564,0.00060332316,0.0013549706,0.0023533066,0.001486449,0.0017272851],"category_scores_gemma":[0.046172325,0.0004900642,0.00129762,0.0016043878,0.00055461016,0.0016707049,0.0020939282,0.0015999615,0.0010767895],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0129682,0.00850689,0.3311955,0.0019769303,0.008837014,0.0004552033,0.0011024543,0.208423,0.02212977,0.0019595346,0.039301246,0.36314428],"study_design_scores_gemma":[0.0012067868,0.0037493054,0.21458757,0.00035359772,0.0017753077,0.00038310772,0.0005286932,0.74135303,0.023924865,0.0034261963,0.008532676,0.00017886952],"about_ca_topic_score_codex":0.0077128606,"about_ca_topic_score_gemma":0.014867001,"teacher_disagreement_score":0.012582537,"about_ca_system_score_codex":0.0011345381,"about_ca_system_score_gemma":0.0013651892,"threshold_uncertainty_score":0.06654364},"labels":[],"label_agreement":null},{"id":"W3082254096","doi":"","title":"Using Deep Learning Classifiers to Identify Candidate Classes for Unit Testing in Object-Oriented Systems.","year":2020,"lang":"en","type":"article","venue":"Software Engineering and Knowledge Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières; Université du Québec","funders":"","keywords":"Computer science; Artificial intelligence; Deep learning; Unit testing; Object (grammar); Machine learning; Unit (ring theory); Pattern recognition (psychology); Mathematics; Programming language; Software","score_opus":0.04895468532458588,"score_gpt":0.29476299749057217,"score_spread":0.24580831216598628,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3082254096","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6547124,0.002199884,0.32797074,0.0013874918,0.00018039365,0.00016591753,0.0008489442,0.0058407704,0.006693434],"genre_scores_gemma":[0.94101727,0.00012252781,0.055631474,0.00019018477,0.0000340104,0.00005536831,0.000992543,0.00010588103,0.0018507703],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99917185,0.0002593773,0.000063914544,0.00017047292,0.0001907079,0.00014365448],"domain_scores_gemma":[0.99286205,0.0050757644,0.0005253181,0.0003697936,0.00091850245,0.00024861834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014163232,0.00084886985,0.00059933675,0.0018130875,0.00041823334,0.001018225,0.0014505369,0.0014065753,0.0015011702],"category_scores_gemma":[0.0071779285,0.0003396204,0.00055306347,0.00071363617,0.00040720764,0.0016725727,0.0007789948,0.0013900648,0.00055177766],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010963506,0.0010005856,0.06483196,0.0002903788,0.00023724539,0.00037882503,0.00023620151,0.21109927,0.014848165,0.00534192,0.0152498325,0.6853892],"study_design_scores_gemma":[0.000018801858,0.000051479743,0.0016009954,0.000017710445,0.000021550066,0.000023957753,0.00002909968,0.9915358,0.0032109183,0.0030372965,0.00044776118,0.000004590077],"about_ca_topic_score_codex":0.010249743,"about_ca_topic_score_gemma":0.014158782,"teacher_disagreement_score":0.010249743,"about_ca_system_score_codex":0.00094615074,"about_ca_system_score_gemma":0.001019754,"threshold_uncertainty_score":0.02038014},"labels":[],"label_agreement":null},{"id":"W3088428890","doi":"10.1145/3387940.3392218","title":"Double Cycle Hybrid Testing of Hybrid Distributed IoT System","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Reliability (semiconductor); Automation; Embedded system; Variety (cybernetics); Internet of Things; Network packet; Packet loss; Software; Reliability engineering; Computer network; Engineering; Operating system","score_opus":0.04061316968890568,"score_gpt":0.24544651359270597,"score_spread":0.20483334390380029,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3088428890","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7539528,0.00018345658,0.23704171,0.00013764815,0.00007312914,0.0001245798,0.0001726307,0.0026300556,0.005683887],"genre_scores_gemma":[0.9873863,0.000012079025,0.011226152,0.00003553969,0.0000035111796,0.000050073188,0.00006026389,0.00004588423,0.0011800667],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987739,0.0003021128,0.000060179416,0.00025258266,0.0004344404,0.00017681607],"domain_scores_gemma":[0.9977575,0.00090615463,0.00017173962,0.0005764152,0.0004530423,0.00013519754],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007717983,0.0005520391,0.00041320975,0.0007200856,0.00030036396,0.0005952337,0.0012197155,0.00059263397,0.0033257757],"category_scores_gemma":[0.0018599018,0.00020785612,0.00030783523,0.00028875755,0.0005625144,0.0011002215,0.0007838921,0.00039214906,0.0002847212],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003925123,0.0006747222,0.025962591,0.00055930135,0.0002543278,0.0034495108,0.0011807376,0.23775277,0.49036303,0.02181794,0.0034842994,0.21057568],"study_design_scores_gemma":[0.000112729926,0.001271331,0.0037317658,0.000024143208,0.0000443111,0.0007009035,0.00013848122,0.8744505,0.10943181,0.00791091,0.0021412414,0.000041876956],"about_ca_topic_score_codex":0.00089945673,"about_ca_topic_score_gemma":0.00086475076,"teacher_disagreement_score":0.0033257757,"about_ca_system_score_codex":0.00049801223,"about_ca_system_score_gemma":0.00036990442,"threshold_uncertainty_score":0.011125863},"labels":[],"label_agreement":null},{"id":"W3092453472","doi":"10.1002/stvr.1751","title":"BUGSJS: a benchmark and taxonomy of JavaScript bugs","year":2020,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"European Social Fund; European Commission; Natural Sciences and Engineering Research Council of Canada; Advanced Remanufacturing and Technology Centre; National Research, Development and Innovation Office; Innovációs és Technológiai Minisztérium","keywords":"JavaScript; Computer science; Unobtrusive JavaScript; Benchmark (surveying); Unit testing; Software bug; Debugging; Taxonomy (biology); Programming language; Web application; Software; Software testing; Test case; Software engineering; Rich Internet application; World Wide Web; Machine learning","score_opus":0.05055143313862166,"score_gpt":0.23784943200647207,"score_spread":0.18729799886785042,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3092453472","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.836064,0.0066206027,0.09647378,0.0011950716,0.00039056546,0.0013581854,0.022664804,0.027221845,0.0080111325],"genre_scores_gemma":[0.78612214,0.0016117098,0.14813918,0.00036205744,0.00011198211,0.0013654436,0.056822624,0.0033294882,0.0021353466],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9837818,0.003243753,0.002657262,0.0018715583,0.0075621326,0.0008834231],"domain_scores_gemma":[0.9476385,0.023495547,0.007570923,0.0058816806,0.013668538,0.0017447505],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007724821,0.0014590021,0.0006202083,0.009900055,0.0010023968,0.0016821609,0.002259229,0.0011249069,0.000761656],"category_scores_gemma":[0.03974611,0.0005094592,0.0009711666,0.005282644,0.0011550979,0.0019736437,0.0022793303,0.001240978,0.0004992518],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015065097,0.0018618307,0.3265269,0.010159151,0.0006476183,0.0020378362,0.0050823567,0.07144265,0.046945777,0.0148598645,0.07948637,0.43944308],"study_design_scores_gemma":[0.0004988544,0.0037590258,0.31582505,0.002588079,0.00053989014,0.0046198107,0.003468318,0.4382162,0.08348332,0.02421858,0.12219081,0.0005920086],"about_ca_topic_score_codex":0.006062353,"about_ca_topic_score_gemma":0.007132469,"teacher_disagreement_score":0.009900055,"about_ca_system_score_codex":0.0013200011,"about_ca_system_score_gemma":0.0022717773,"threshold_uncertainty_score":0.040853262},"labels":[],"label_agreement":null},{"id":"W3095252018","doi":"10.1109/tse.2021.3070549","title":"Reinforcement Learning for Test Case Prioritization","year":2021,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Queen's University; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Reinforcement learning; Computer science; Prioritization; Context (archaeology); Ranking (information retrieval); Regression testing; Machine learning; Test case; Bidding; Test (biology); Adaptation (eye); Artificial intelligence; Regression analysis; Engineering; Software; Software system","score_opus":0.019405390937673923,"score_gpt":0.25133358157887337,"score_spread":0.23192819064119943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3095252018","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044530284,0.0005961345,0.950491,0.00042248267,0.000050715906,0.0002295436,0.00005841336,0.0014120727,0.002209305],"genre_scores_gemma":[0.8594105,0.0002341544,0.1380757,0.00023519936,0.00005395097,0.00029360692,0.00016989782,0.00011648532,0.0014103882],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963223,0.0017049817,0.00019311353,0.000692643,0.00071792403,0.00036898546],"domain_scores_gemma":[0.98490983,0.011297194,0.001338123,0.00076456415,0.0010951532,0.0005950878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050872313,0.0016333914,0.0013696098,0.0011302733,0.00045552646,0.0012069248,0.0021645795,0.001032445,0.0025040274],"category_scores_gemma":[0.021886146,0.00067581545,0.00066953036,0.00064972515,0.0013361719,0.0015663966,0.0012729962,0.0027242738,0.00047004188],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001974176,0.00029238602,0.00279596,0.00017980649,0.00008751017,0.00008833702,0.00011635881,0.8632915,0.0027415934,0.005763075,0.0011566782,0.123289436],"study_design_scores_gemma":[0.000024587447,0.00007101423,0.00020471972,0.000010811644,0.0000135920045,0.000016023605,0.000008640164,0.99583274,0.00065345987,0.0028764207,0.00028149126,0.0000065239597],"about_ca_topic_score_codex":0.0055242903,"about_ca_topic_score_gemma":0.0058353418,"teacher_disagreement_score":0.0055242903,"about_ca_system_score_codex":0.0017969427,"about_ca_system_score_gemma":0.0025833086,"threshold_uncertainty_score":0.026904166},"labels":[],"label_agreement":null},{"id":"W3096465442","doi":"10.1002/spe.2929","title":"Root causing, detecting, and fixing flaky tests: State of the art and future roadmap","year":2020,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brandon University; University of Guelph; University of Guelph-Humber","funders":"","keywords":"Pace; Root cause; Software deployment; Test (biology); Field (mathematics); Computer science; Root (linguistics); Engineering; Data science; Software; Engineering management; Software engineering; Operations management","score_opus":0.016750706809614355,"score_gpt":0.2725751747409247,"score_spread":0.2558244679313103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3096465442","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021416496,0.97667706,0.011913887,0.005028501,0.00047131753,0.00010237458,0.000104464445,0.0002280186,0.003332792],"genre_scores_gemma":[0.030990623,0.93528235,0.028969929,0.002194962,0.00075502135,0.00018437843,0.0004171135,0.00007140089,0.0011342815],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99255365,0.0023103775,0.000916176,0.0012380532,0.002572442,0.000409303],"domain_scores_gemma":[0.92818105,0.04969717,0.0043386663,0.0018842074,0.014790902,0.0011079168],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01412195,0.0016826863,0.0017715236,0.009353181,0.0007912846,0.004579197,0.0035757448,0.0035283642,0.005459324],"category_scores_gemma":[0.025605418,0.00077619904,0.0019500157,0.0055131237,0.002280566,0.009019743,0.0018547882,0.002737615,0.0017489251],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000119121185,0.00022339761,0.0028308204,0.018246338,0.000105010055,0.000091527654,0.00037396883,0.0015105571,0.0011932233,0.01000656,0.008221201,0.9570782],"study_design_scores_gemma":[0.0001109489,0.002025121,0.0121989,0.10729045,0.0020738433,0.0020928294,0.0063292817,0.019268775,0.008748109,0.060534377,0.77890533,0.0004219422],"about_ca_topic_score_codex":0.0050029587,"about_ca_topic_score_gemma":0.0038488663,"teacher_disagreement_score":0.01412195,"about_ca_system_score_codex":0.0021289142,"about_ca_system_score_gemma":0.0076879645,"threshold_uncertainty_score":0.07468492},"labels":[],"label_agreement":null},{"id":"W3097853154","doi":"10.1109/icsme46990.2020.00047","title":"A Cost-Effective Approach for Hyper-Parameter Tuning in Search-based Test Case Generation","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Metric (unit); Computer science; Heuristic; Fine-tuning; Class (philosophy); Domain (mathematical analysis); Selection (genetic algorithm); Mathematical optimization; Test case; Machine learning; Artificial intelligence; Mathematics; Engineering","score_opus":0.15115330699758797,"score_gpt":0.32558085003198906,"score_spread":0.17442754303440108,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3097853154","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.069067724,0.0009367747,0.9219697,0.0004111215,0.00006156642,0.00055923744,0.00021484958,0.0033503422,0.0034285246],"genre_scores_gemma":[0.57845503,0.00022985013,0.41874757,0.00020717666,0.000032383243,0.0006466191,0.00044364604,0.00041187325,0.00082583295],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99487114,0.0023085582,0.00031771595,0.00063316355,0.0015319425,0.00033752012],"domain_scores_gemma":[0.9861893,0.009278324,0.0010814288,0.0015827478,0.0016139679,0.00025411882],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004168613,0.0021954724,0.0013446913,0.0040657795,0.000537492,0.0013814685,0.0023553532,0.0017232242,0.0031297896],"category_scores_gemma":[0.023910765,0.00088297896,0.001331935,0.002167491,0.00090588664,0.0018391195,0.0014394802,0.001579682,0.00055917184],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041691784,0.00041438962,0.008004238,0.00038940707,0.00027398052,0.0001815753,0.00020129341,0.6271959,0.016099053,0.007926347,0.0023291253,0.33656773],"study_design_scores_gemma":[0.000062929,0.00013772106,0.001152702,0.000036921265,0.00008726875,0.00010199093,0.000037359547,0.98938656,0.0043420764,0.003794406,0.0008375623,0.000022437169],"about_ca_topic_score_codex":0.003002464,"about_ca_topic_score_gemma":0.0035343224,"teacher_disagreement_score":0.004168613,"about_ca_system_score_codex":0.001585214,"about_ca_system_score_gemma":0.0020467434,"threshold_uncertainty_score":0.02204603},"labels":[],"label_agreement":null},{"id":"W3098557859","doi":"10.1145/3368089.3409757","title":"ARDiff: scaling program equivalence checking via iterative abstraction and refinement of common code","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Equivalence (formal languages); Symbolic execution; Computer science; Symbolic trajectory evaluation; Abstraction; Programming language; Program analysis; Formal equivalence checking; Abstraction model checking; Model checking; Theoretical computer science; Symbolic data analysis; Static analysis; Algorithm; Mathematics; Discrete mathematics; Software","score_opus":0.05157918219343462,"score_gpt":0.31744576279154774,"score_spread":0.26586658059811313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3098557859","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008780707,0.00035352193,0.970327,0.00017655778,0.000118559146,0.00028869818,0.00021384652,0.017097007,0.0026441133],"genre_scores_gemma":[0.16784869,0.0003384892,0.8192948,0.00044876285,0.00009583166,0.00059618795,0.0017584276,0.006118656,0.0035001738],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97852933,0.006678573,0.0013922905,0.0033555448,0.008717801,0.0013264023],"domain_scores_gemma":[0.96473974,0.012457208,0.001507614,0.016122263,0.004652713,0.0005205032],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00986147,0.0022453011,0.002013061,0.004196802,0.0014836349,0.0029083774,0.0060402635,0.001896013,0.0084680375],"category_scores_gemma":[0.04088115,0.0015868336,0.0043379497,0.002332539,0.0036580905,0.008187623,0.010607688,0.005245195,0.003317409],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008940438,0.0007507177,0.008142159,0.0013911849,0.0005510164,0.000705107,0.0014401422,0.10558761,0.037686538,0.16798154,0.018309755,0.6565602],"study_design_scores_gemma":[0.00031029148,0.00050271506,0.0015649087,0.00037554116,0.00028252308,0.000444887,0.00030123236,0.68691826,0.03857138,0.23776479,0.032792088,0.00017145158],"about_ca_topic_score_codex":0.0065218997,"about_ca_topic_score_gemma":0.00782364,"teacher_disagreement_score":0.00986147,"about_ca_system_score_codex":0.0019342684,"about_ca_system_score_gemma":0.0041044876,"threshold_uncertainty_score":0.05215305},"labels":[],"label_agreement":null},{"id":"W3100648313","doi":"10.1145/3412452.3423571","title":"Comparing transition trees test suites effectiveness for different mutation operators","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Mutation testing; Computer science; Mutation; Fault tree analysis; Test case; Process (computing); Test strategy; Tree (set theory); Fault (geology); Reliability engineering; Data mining; Artificial intelligence; Machine learning; Engineering; Mathematics; Programming language; Biology","score_opus":0.040456044367176924,"score_gpt":0.27180922118318696,"score_spread":0.23135317681601003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3100648313","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98326427,0.00064412336,0.013597205,0.000105103834,0.000024908035,0.0000750581,0.00027682248,0.00088822335,0.0011242295],"genre_scores_gemma":[0.98154676,0.00025839452,0.0169614,0.00003463488,0.000009997195,0.00006900012,0.00072672905,0.00009676515,0.00029636154],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956873,0.0016637888,0.0006271183,0.00048364096,0.0012704475,0.00026778082],"domain_scores_gemma":[0.94912964,0.040256444,0.0033566116,0.003105411,0.0034510852,0.00070088514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048399614,0.00079405593,0.0005994844,0.003256343,0.0003059921,0.00076848513,0.00079051085,0.0008094855,0.00056303234],"category_scores_gemma":[0.03338626,0.00024680022,0.00058398413,0.001497418,0.00062601455,0.0010834762,0.000540405,0.0005464293,0.00012999686],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034402441,0.0018906525,0.10412953,0.0011035033,0.0012614138,0.00069892,0.00058476755,0.3886722,0.12629369,0.004153377,0.002302648,0.36546904],"study_design_scores_gemma":[0.00042575048,0.010141323,0.087882765,0.0001383137,0.0007346311,0.0017671529,0.000616702,0.72684294,0.16471483,0.0035829104,0.0029505051,0.00020210381],"about_ca_topic_score_codex":0.0015164908,"about_ca_topic_score_gemma":0.00296777,"teacher_disagreement_score":0.0048399614,"about_ca_system_score_codex":0.00072138367,"about_ca_system_score_gemma":0.0006068179,"threshold_uncertainty_score":0.02559644},"labels":[],"label_agreement":null},{"id":"W3102206389","doi":"10.1145/3419604.3419628","title":"Test Generation Tool for Modified Condition/Decision Coverage","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Traceability; Model-based testing; Integration testing; Code coverage; Test strategy; White-box testing; Dataflow; Extended finite-state machine; Non-regression testing; Keyword-driven testing; Manual testing; Reliability engineering; Software performance testing; Test case; Software; Finite-state machine; Software system; Algorithm; Software engineering; Programming language; Engineering; Software construction; Machine learning","score_opus":0.05601324452544399,"score_gpt":0.29192293129835284,"score_spread":0.23590968677290886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3102206389","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007928079,0.00006127276,0.9387729,0.00010304199,0.00004389386,0.00021447775,0.0010236016,0.046932854,0.004919906],"genre_scores_gemma":[0.20248412,0.0001319984,0.7776916,0.0002354639,0.000046784353,0.00093049655,0.006605626,0.0060379677,0.00583593],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985771,0.00035825543,0.0001456614,0.00024722228,0.0005491393,0.00012256634],"domain_scores_gemma":[0.99729,0.0017891953,0.00014651622,0.00034375096,0.00038958844,0.00004094876],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011990188,0.0012725549,0.00056914956,0.0023216687,0.0003440349,0.0010405623,0.0015296034,0.0012880266,0.019084059],"category_scores_gemma":[0.006289598,0.00051322364,0.0012338196,0.0010424545,0.0005914549,0.0011746382,0.0010038879,0.001085345,0.0031516806],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009210097,0.0005715498,0.004904726,0.0011169749,0.00014458623,0.0023380814,0.00051544054,0.12720908,0.07497027,0.090285785,0.048880253,0.6481424],"study_design_scores_gemma":[0.00031814314,0.00026373737,0.0008991444,0.00012949822,0.000073241754,0.0012138677,0.000046160894,0.83597517,0.081894,0.031547412,0.04757065,0.000069054186],"about_ca_topic_score_codex":0.0015929335,"about_ca_topic_score_gemma":0.0010399118,"teacher_disagreement_score":0.019084059,"about_ca_system_score_codex":0.00059215917,"about_ca_system_score_gemma":0.00086343545,"threshold_uncertainty_score":0.063842535},"labels":[],"label_agreement":null},{"id":"W3105178636","doi":"10.1145/3368089.3409737","title":"Mining assumptions for software components using machine learning","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg; European Commission; H2020 European Research Council; Université du Luxembourg","keywords":"Computer science; Component (thermodynamics); Set (abstract data type); Machine learning; Test case; Software; Decision tree; Process (computing); Tree (set theory); Artificial intelligence; Component-based software engineering; Software system; Data mining; Programming language; Mathematics","score_opus":0.12612852738100291,"score_gpt":0.3063556652720434,"score_spread":0.18022713789104047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3105178636","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18846343,0.0007439703,0.7978871,0.00064204197,0.000049296235,0.000350113,0.0024796987,0.0075857583,0.0017986395],"genre_scores_gemma":[0.65346116,0.00027952873,0.33871937,0.00015858318,0.00003719639,0.0002772052,0.0062664547,0.00027744655,0.00052299246],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9945913,0.0012479494,0.00064017606,0.0012693777,0.0019084078,0.0003427792],"domain_scores_gemma":[0.97335804,0.019712226,0.0020785304,0.00223667,0.002337202,0.00027728887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030725538,0.0020250925,0.0009940735,0.0054799244,0.0008713731,0.0019051388,0.0029008866,0.0012345273,0.0019261295],"category_scores_gemma":[0.03138533,0.00093581865,0.0027131566,0.0017187978,0.0013476927,0.0037511382,0.0020853975,0.0017698118,0.00066404504],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00095419644,0.00038640975,0.065055266,0.0011142789,0.00041073895,0.0022690399,0.00093585043,0.52444243,0.014769571,0.01813371,0.005727967,0.36580053],"study_design_scores_gemma":[0.000049014383,0.000120301724,0.0026713433,0.0001030307,0.000084280284,0.00022339328,0.00020480531,0.9655793,0.008116833,0.021045052,0.0017729164,0.000029708008],"about_ca_topic_score_codex":0.0049191853,"about_ca_topic_score_gemma":0.0077159707,"teacher_disagreement_score":0.0054799244,"about_ca_system_score_codex":0.0017038941,"about_ca_system_score_gemma":0.0026630193,"threshold_uncertainty_score":0.016249418},"labels":[],"label_agreement":null},{"id":"W3108741386","doi":"10.1145/3417564.3417575","title":"Genetic Improvement @ ICSE 2020","year":2020,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Engineering and Physical Sciences Research Council","keywords":"Coronavirus disease 2019 (COVID-19); Face (sociological concept); 2019-20 coronavirus outbreak; Severe acute respiratory syndrome coronavirus 2 (SARS-CoV-2); The Internet; Personal account; Software; Computer science; Pandemic; World Wide Web; Sociology; Medicine; Art; Virology; Operating system; Social science","score_opus":0.01721808277170105,"score_gpt":0.22137228675128495,"score_spread":0.2041542039795839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3108741386","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005968052,0.0043202215,0.021984037,0.034951795,0.009945582,0.0002826312,0.0055433875,0.01569835,0.901306],"genre_scores_gemma":[0.012929803,0.0018071802,0.010879827,0.004132639,0.0006615395,0.00011549887,0.0037233057,0.0015175403,0.96423274],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991918,0.00013359266,0.000024078496,0.00013670101,0.00039198395,0.00012178074],"domain_scores_gemma":[0.9982358,0.00024721984,0.00008336611,0.0002736419,0.00047001988,0.00068988506],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0021033238,0.0010804964,0.00036544612,0.0011053471,0.00082811585,0.002601428,0.0005544799,0.0017121345,0.40568766],"category_scores_gemma":[0.0026276393,0.00026744994,0.0004496267,0.0009146054,0.0005132635,0.0013816962,0.0012950308,0.0022923835,0.19111788],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000093214556,0.00011887899,0.00047345614,0.00008110331,0.000006907667,0.00013853925,0.000051110885,0.00026992906,0.003345797,0.011228746,0.8048637,0.17932858],"study_design_scores_gemma":[0.00001737626,0.00007372465,0.00045509957,0.000030742976,0.0000028132954,0.000089624074,0.000034025736,0.00029631692,0.0012137285,0.0023351717,0.99544394,0.0000074432132],"about_ca_topic_score_codex":0.0012767123,"about_ca_topic_score_gemma":0.0025918137,"teacher_disagreement_score":0.40568766,"about_ca_system_score_codex":0.0010730876,"about_ca_system_score_gemma":0.0014366098,"threshold_uncertainty_score":0.8477144},"labels":[],"label_agreement":null},{"id":"W3110815134","doi":"10.5381/jot.2021.20.3.a9","title":"Fixing Multiple Type Errors in Model Transformations with Alternative Oracles to Test Cases.","year":2021,"lang":"en","type":"preprint","venue":"The Journal of Object Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Institut de Valorisation des Données; Canada First Research Excellence Fund","keywords":"Heuristics; Transformation (genetics); Computer science; Type I and type II errors; Type (biology); Model transformation; Algorithm; Selection (genetic algorithm); Space (punctuation); Artificial intelligence; Theoretical computer science; Machine learning; Mathematics; Statistics","score_opus":0.03467816425871621,"score_gpt":0.29206771445440244,"score_spread":0.2573895501956862,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3110815134","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1682242,0.0005592566,0.8206618,0.0003383007,0.00006023485,0.00032928854,0.00013300088,0.006485248,0.0032086151],"genre_scores_gemma":[0.6747874,0.00016171922,0.3222301,0.00018845864,0.000023327344,0.00013071726,0.00031128613,0.00081792625,0.0013491036],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99336845,0.0027331493,0.00035861938,0.0010581417,0.0020524706,0.00042921322],"domain_scores_gemma":[0.9711845,0.01672426,0.0029057264,0.0073974477,0.0013840253,0.00040405645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051308973,0.0012992679,0.00087907043,0.0015716853,0.0003894774,0.0013659553,0.0024848236,0.0013893018,0.002087705],"category_scores_gemma":[0.03717124,0.00065260875,0.0011618469,0.00086153834,0.0015817522,0.0021042558,0.0020991985,0.0017182064,0.00052352564],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094569754,0.0009584187,0.03278214,0.0007729066,0.0003183711,0.0013622728,0.0012418542,0.31444252,0.04515282,0.015245128,0.0027393491,0.58403856],"study_design_scores_gemma":[0.00024338187,0.001604725,0.007758153,0.0003035764,0.0003854689,0.0020737157,0.00052634947,0.88518006,0.059311662,0.029622853,0.012890757,0.00009934024],"about_ca_topic_score_codex":0.0016325611,"about_ca_topic_score_gemma":0.0027984204,"teacher_disagreement_score":0.0051308973,"about_ca_system_score_codex":0.0007108285,"about_ca_system_score_gemma":0.0014257637,"threshold_uncertainty_score":0.027135134},"labels":[],"label_agreement":null},{"id":"W3118132313","doi":"10.1109/ictai50040.2020.00091","title":"Decision Support for Combining Security Mechanisms using Exploratory Evolutionary Testing","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Decision support system; Artificial intelligence","score_opus":0.0980905697129831,"score_gpt":0.3002942782632393,"score_spread":0.20220370855025618,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3118132313","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15971176,0.00018071635,0.8303616,0.000686541,0.00002660332,0.0006861293,0.00022884288,0.002689944,0.005427743],"genre_scores_gemma":[0.5224134,0.00006735547,0.47581545,0.00017451531,0.000018250306,0.00041192645,0.00031876474,0.00010273482,0.00067762093],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9922672,0.0047213975,0.0004106299,0.00067541236,0.0015167993,0.00040858457],"domain_scores_gemma":[0.95661604,0.037692748,0.0015596173,0.0014683448,0.0021323662,0.0005309257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01033038,0.0020649119,0.0011270784,0.0035770568,0.00080499053,0.002375757,0.0035661533,0.0019091917,0.0055240337],"category_scores_gemma":[0.045031358,0.00094721466,0.0015668746,0.0011721328,0.0011601648,0.0025409115,0.0020940467,0.0013171774,0.0005785046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008627803,0.0016227426,0.026253788,0.000538912,0.00041683944,0.0009983155,0.000954865,0.68734556,0.011447561,0.017142741,0.0019196139,0.2504963],"study_design_scores_gemma":[0.00009171739,0.0003552544,0.00058034965,0.000048401136,0.0000584282,0.00011399651,0.00014065245,0.9870077,0.0029173186,0.007900704,0.0007565538,0.000029012237],"about_ca_topic_score_codex":0.0024511607,"about_ca_topic_score_gemma":0.003730497,"teacher_disagreement_score":0.01033038,"about_ca_system_score_codex":0.0010027841,"about_ca_system_score_gemma":0.0018041271,"threshold_uncertainty_score":0.054632902},"labels":[],"label_agreement":null},{"id":"W3119230199","doi":"10.1109/tse.2021.3107680","title":"Mutation Analysis for Cyber-Physical Systems: Scalable Solutions and Results in the Space Domain","year":2021,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"H2020 European Research Council; Natural Sciences and Engineering Research Council of Canada; European Commission; European Space Agency","keywords":"Computer science; Context (archaeology); Software construction; Software; Software system; Scalability; Software reliability testing; Software engineering; Verification and validation; Avionics software; Domain (mathematical analysis); Reliability engineering; Engineering; Operating system","score_opus":0.02396113930530181,"score_gpt":0.25378680370064594,"score_spread":0.22982566439534413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119230199","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33464003,0.012757143,0.61867094,0.0051562344,0.0004506228,0.00040547756,0.00037197678,0.0041952203,0.02335239],"genre_scores_gemma":[0.69123286,0.00522077,0.29603848,0.0005420226,0.00024717324,0.0002792966,0.00059174077,0.0005546261,0.0052929507],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979359,0.0007822201,0.000070536094,0.00023739711,0.00073950033,0.0002344591],"domain_scores_gemma":[0.9876236,0.009811103,0.00037879046,0.0007904872,0.0009967806,0.0003993597],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037042864,0.0013080294,0.0013135675,0.0013923435,0.0005891883,0.0014108898,0.0012627202,0.0016591485,0.0031096896],"category_scores_gemma":[0.014430895,0.00026548607,0.00161687,0.0013828692,0.0016272615,0.0028933615,0.0018324082,0.0025743283,0.0004911979],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069382525,0.0010990166,0.0040915534,0.00069023034,0.00022322788,0.00055123144,0.00025554313,0.6278932,0.015396129,0.06251852,0.007388059,0.27919942],"study_design_scores_gemma":[0.00010329026,0.0001962132,0.0009554551,0.000050805258,0.00004702004,0.00007508358,0.00008325354,0.9665072,0.0049331854,0.025325196,0.0017030936,0.00002018946],"about_ca_topic_score_codex":0.0050728964,"about_ca_topic_score_gemma":0.0029388098,"teacher_disagreement_score":0.0050728964,"about_ca_system_score_codex":0.0013824352,"about_ca_system_score_gemma":0.0013048378,"threshold_uncertainty_score":0.019590318},"labels":[],"label_agreement":null},{"id":"W31287713","doi":"10.1007/978-3-642-31753-8_18","title":"JSART: JavaScript Assertion-Based Regression Testing","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"JavaScript; Unobtrusive JavaScript; Computer science; Programming language; Correctness; Regression testing; Assertion; Rich Internet application; Source code; Test case; Web application; Data mining; Artificial intelligence; Machine learning; Operating system; Regression analysis; Software; Software development","score_opus":0.04071869676340591,"score_gpt":0.2725068548380246,"score_spread":0.23178815807461872,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W31287713","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0046458365,0.00028830976,0.87469196,0.00017378542,0.00020058511,0.00008835494,0.000509914,0.112243086,0.0071582547],"genre_scores_gemma":[0.25121418,0.00074823946,0.6708354,0.0007578321,0.00025928137,0.00033651982,0.004270162,0.044176687,0.027401753],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970306,0.0006068257,0.0002184356,0.00043144828,0.0015121454,0.00020040666],"domain_scores_gemma":[0.9951781,0.0025528271,0.00032944497,0.001134554,0.00066173036,0.00014334607],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002181398,0.001982767,0.0010156551,0.0013908566,0.0004484231,0.0015301803,0.0035746982,0.0013574447,0.015927985],"category_scores_gemma":[0.007294555,0.0010384257,0.0012311325,0.0008745598,0.0010635235,0.00279632,0.0017366781,0.0025149612,0.0068248473],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008482565,0.0005128182,0.0028020483,0.001023424,0.00017485449,0.0009202764,0.0003860093,0.030277537,0.059944373,0.040682487,0.08739787,0.77503],"study_design_scores_gemma":[0.00036659656,0.00044658856,0.0027159406,0.00047213215,0.00029247746,0.00203711,0.000087935405,0.61716074,0.15574838,0.092926376,0.12752475,0.00022102002],"about_ca_topic_score_codex":0.0012863722,"about_ca_topic_score_gemma":0.0012540204,"teacher_disagreement_score":0.015927985,"about_ca_system_score_codex":0.00034604044,"about_ca_system_score_gemma":0.0007581233,"threshold_uncertainty_score":0.053284466},"labels":[],"label_agreement":null},{"id":"W3131260619","doi":"10.1109/sbst52555.2021.00010","title":"Data Driven Testing of Cyber Physical Systems","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Testbed; Computer science; Python (programming language); Cyber-physical system; Thermostat; Software deployment; Software; Task (project management); Software engineering; Domain (mathematical analysis); Implementation; Distributed computing; Systems engineering; Engineering; Programming language","score_opus":0.1260824671765509,"score_gpt":0.32881512688358794,"score_spread":0.20273265970703705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3131260619","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16940047,0.0003579414,0.81951046,0.00044011863,0.0000735437,0.00016641687,0.0007678207,0.0056969468,0.0035863023],"genre_scores_gemma":[0.8770521,0.00017208894,0.1202589,0.00011332092,0.000017809049,0.00025902735,0.0008916212,0.00034912163,0.00088606525],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99641883,0.0014129917,0.0001916289,0.00046588012,0.0012754512,0.00023517992],"domain_scores_gemma":[0.9895263,0.007350687,0.000682182,0.0014230129,0.0008083647,0.00020953422],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024261118,0.0009916641,0.0005447457,0.0010853161,0.00028841718,0.0012207365,0.0017843796,0.00096557464,0.0020402963],"category_scores_gemma":[0.013737449,0.000498659,0.0008390925,0.0007067483,0.001832322,0.0017115403,0.0012490994,0.00097619265,0.00028121244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003116287,0.00020868657,0.0064130924,0.0004763138,0.00007585002,0.00042735192,0.0002646444,0.8915849,0.014578928,0.021220608,0.0013071762,0.063130885],"study_design_scores_gemma":[0.000025800142,0.00006345935,0.00056598877,0.000024300774,0.000007989981,0.00006821667,0.00002236102,0.97275734,0.011717064,0.013726385,0.0010111495,0.000010044559],"about_ca_topic_score_codex":0.0017756815,"about_ca_topic_score_gemma":0.0013757083,"teacher_disagreement_score":0.0024261118,"about_ca_system_score_codex":0.00094651233,"about_ca_system_score_gemma":0.0010873257,"threshold_uncertainty_score":0.012830675},"labels":[],"label_agreement":null},{"id":"W3135281973","doi":"10.32628/cseit1949168","title":"Impact of Automation on the Test Insertion","year":2019,"lang":"en","type":"article","venue":"International Journal of Scientific Research in Computer Science Engineering and Information Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Automation; Test (biology); Computer science; Engineering; Geology; Mechanical engineering","score_opus":0.02411700522098179,"score_gpt":0.32688362792859393,"score_spread":0.3027666227076121,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3135281973","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92493945,0.001604767,0.049246285,0.0009405767,0.00038616877,0.00008794442,0.00076600676,0.009966051,0.012062798],"genre_scores_gemma":[0.98958313,0.00013123553,0.008378885,0.0000973625,0.00008112374,0.000015143159,0.00034752258,0.0003410831,0.0010244387],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98669064,0.004497534,0.0007320751,0.0014026762,0.005245639,0.001431427],"domain_scores_gemma":[0.80756396,0.14772101,0.0077433833,0.020441627,0.014797796,0.0017321957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003651954,0.0012886511,0.0008026081,0.0022638699,0.0007999826,0.00244018,0.0016110974,0.0017235571,0.008848095],"category_scores_gemma":[0.07455724,0.0005101901,0.0010928556,0.0018052341,0.0008473029,0.0027378155,0.0012789005,0.0015575398,0.0021226408],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013773024,0.0027469317,0.120240286,0.0014847044,0.000588316,0.0029062922,0.0014230385,0.17735465,0.116548836,0.0074359938,0.007959281,0.5475387],"study_design_scores_gemma":[0.0005561675,0.009752358,0.24999167,0.00036371194,0.0021586376,0.00593532,0.0014450618,0.51455325,0.18601525,0.01025351,0.01861928,0.0003558223],"about_ca_topic_score_codex":0.002324842,"about_ca_topic_score_gemma":0.0013289575,"teacher_disagreement_score":0.008848095,"about_ca_system_score_codex":0.0006584787,"about_ca_system_score_gemma":0.0015375657,"threshold_uncertainty_score":0.029599845},"labels":[],"label_agreement":null},{"id":"W3142089163","doi":"10.1109/ase.2004.1342761","title":"Using a genetic algorithm and formal concept analysis to generate branch coverage test data automatically","year":2004,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Pointer (user interface); Algorithm; Pairwise comparison; Test suite; Theoretical computer science; Programming language; Artificial intelligence; Test case; Machine learning","score_opus":0.04303292143597743,"score_gpt":0.30504203559225707,"score_spread":0.26200911415627964,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3142089163","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019432105,0.00004180406,0.9757706,0.00011363329,0.000019631963,0.00015488666,0.00009350143,0.0018705486,0.002503215],"genre_scores_gemma":[0.13492586,0.000079889556,0.8625544,0.0001009085,0.00001462311,0.0003740977,0.00039806316,0.00027147232,0.0012806446],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986344,0.00036651525,0.00006783014,0.00024409426,0.00057302625,0.00011411887],"domain_scores_gemma":[0.9963987,0.0026214777,0.00018086929,0.00023357077,0.0005207324,0.00004458971],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013247897,0.0010398419,0.00066539814,0.002618209,0.0005557267,0.0009818615,0.0012473357,0.0008377883,0.0031879249],"category_scores_gemma":[0.008160462,0.00047065658,0.0013939588,0.0012021792,0.0015963742,0.0010669032,0.0007771912,0.00089046854,0.00051578804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000089103894,0.00015584026,0.0029776993,0.00025442214,0.00006579206,0.00037122564,0.00035338753,0.6074093,0.010209115,0.06764172,0.0030691484,0.3074032],"study_design_scores_gemma":[0.000048182672,0.000055973476,0.00031760617,0.000037019607,0.000022290438,0.000100144534,0.00004436028,0.9638342,0.00445565,0.028504424,0.0025635827,0.000016492182],"about_ca_topic_score_codex":0.007374679,"about_ca_topic_score_gemma":0.0056868047,"teacher_disagreement_score":0.007374679,"about_ca_system_score_codex":0.0014014882,"about_ca_system_score_gemma":0.0024211903,"threshold_uncertainty_score":0.0146635175},"labels":[],"label_agreement":null},{"id":"W3143871025","doi":"10.1007/s00165-005-0082-8","title":"A formal approach to property testing in causally consistent distributed traces","year":2006,"lang":"en","type":"article","venue":"Formal Aspects of Computing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"TRACE (psycholinguistics); Computer science; Theory of computation; Property (philosophy); Automaton; Consistency (knowledge bases); Model checking; Causality (physics); Theoretical computer science; Programming language; Property testing; Event (particle physics); Relation (database); Algorithm; Data mining; Artificial intelligence","score_opus":0.025169513752820116,"score_gpt":0.2376320665976636,"score_spread":0.21246255284484347,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3143871025","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008203975,0.000036613776,0.9982887,0.000108446315,0.000011303824,0.000050287897,0.000026730811,0.00023052907,0.0004269354],"genre_scores_gemma":[0.08939059,0.00023423141,0.90792,0.0001581667,0.000093942326,0.000505062,0.00018668076,0.0001583336,0.0013529134],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.992644,0.0026628596,0.0007065754,0.0009943468,0.0024419688,0.0005502697],"domain_scores_gemma":[0.9809656,0.012734949,0.0013049239,0.0027647507,0.0018446725,0.00038507977],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008665362,0.0013833487,0.0010289081,0.0028684048,0.0017186401,0.0040071234,0.004454415,0.0019588917,0.0037770437],"category_scores_gemma":[0.021948954,0.0012159188,0.0026369246,0.0022696662,0.008567495,0.0069828616,0.0033585746,0.0048007523,0.00057824055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000032148084,0.00012478298,0.0006502703,0.00015403291,0.000038217448,0.0003736208,0.0006025088,0.06016042,0.0033474118,0.91186947,0.00062022544,0.02202689],"study_design_scores_gemma":[0.00004912179,0.00009526277,0.00018703092,0.00015931192,0.000051241877,0.00035244488,0.00016626598,0.35981658,0.0063949144,0.61710864,0.015561826,0.00005729036],"about_ca_topic_score_codex":0.0045657344,"about_ca_topic_score_gemma":0.0038988595,"teacher_disagreement_score":0.008665362,"about_ca_system_score_codex":0.0027732342,"about_ca_system_score_gemma":0.004768277,"threshold_uncertainty_score":0.04582739},"labels":[],"label_agreement":null},{"id":"W3149550336","doi":"10.1109/raise.2012.6227969","title":"Predicting mutation score using source code and test suite metrics","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Test suite; Computer science; Mutation; Mutation testing; Source code; Suite; Code coverage; Test case; Process (computing); Code (set theory); Reliability engineering; Programming language; Machine learning; Software; Set (abstract data type); Engineering; Biology; Genetics","score_opus":0.05832407227421543,"score_gpt":0.2908110474789894,"score_spread":0.23248697520477396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3149550336","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8203655,0.0004456964,0.1724551,0.00027778072,0.000032875865,0.000116711344,0.0014768442,0.0036220197,0.0012074948],"genre_scores_gemma":[0.9305713,0.00014010486,0.066116005,0.000033810127,0.000025701598,0.00006714181,0.0025888784,0.000099137986,0.00035789673],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99814916,0.000438039,0.00017710195,0.00031069588,0.0008036841,0.00012137171],"domain_scores_gemma":[0.9821034,0.0097236475,0.0028396873,0.0008206629,0.0039176177,0.0005949507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002015788,0.0013494613,0.0008296457,0.007791116,0.00022441104,0.00092218677,0.0006934564,0.0013099124,0.00051684026],"category_scores_gemma":[0.020483164,0.00027457584,0.00065726956,0.0027340504,0.0002890911,0.0013276399,0.00045466638,0.0006501203,0.00046373342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028149068,0.0006826024,0.4073517,0.00014775356,0.00027338578,0.00040494127,0.000069298,0.2968413,0.01802536,0.0009083126,0.002524319,0.27248958],"study_design_scores_gemma":[0.000011434064,0.00012670964,0.02972731,0.000009666934,0.00003108952,0.00013040594,0.000014529262,0.96321183,0.005472694,0.0010135934,0.0002304443,0.000020232108],"about_ca_topic_score_codex":0.0035101122,"about_ca_topic_score_gemma":0.00399522,"teacher_disagreement_score":0.007791116,"about_ca_system_score_codex":0.0006228624,"about_ca_system_score_gemma":0.0005854091,"threshold_uncertainty_score":0.010660589},"labels":[],"label_agreement":null},{"id":"W3150233949","doi":"10.1109/date.2012.6176548","title":"Non-solution implications using reverse domination in a modern SAT-based debugging environment","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Debugging; Pruning; Satisfiability modulo theories; Leverage (statistics); Bottleneck; Solver; Speedup; Parallel computing; Algorithmic program debugging; Theoretical computer science; Programming language; Embedded system; Artificial intelligence","score_opus":0.03952488856495533,"score_gpt":0.2789340241394563,"score_spread":0.23940913557450094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3150233949","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028165504,0.0002030354,0.9668841,0.00039820437,0.000018477074,0.00008521173,0.000047894904,0.0023797527,0.0018177965],"genre_scores_gemma":[0.21570201,0.00020711913,0.7822088,0.00019933388,0.000027879496,0.0001014564,0.00012402267,0.00023527513,0.0011941103],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99709463,0.0016283595,0.00012771873,0.00032730435,0.0006565052,0.0001654074],"domain_scores_gemma":[0.99211395,0.005989555,0.00045558697,0.00088873366,0.00044268864,0.00010950408],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002682997,0.0007849243,0.00064620917,0.00144995,0.00062941416,0.0013981728,0.0016118683,0.0009038838,0.0032228793],"category_scores_gemma":[0.009032657,0.00059682544,0.0008382022,0.0011023927,0.0013731982,0.0032707169,0.0018922538,0.0016759193,0.00040920588],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005929199,0.00041141888,0.004506986,0.0006087095,0.000094246636,0.001056543,0.0011409604,0.1532569,0.057351306,0.14302081,0.0044859718,0.6334733],"study_design_scores_gemma":[0.00017533329,0.00023936671,0.00069182576,0.00008320492,0.00007655976,0.0007489475,0.00012900002,0.83558315,0.035321735,0.117898084,0.009002398,0.00005050697],"about_ca_topic_score_codex":0.0010852189,"about_ca_topic_score_gemma":0.0025878942,"teacher_disagreement_score":0.0032228793,"about_ca_system_score_codex":0.0005745164,"about_ca_system_score_gemma":0.001424783,"threshold_uncertainty_score":0.014189243},"labels":[],"label_agreement":null},{"id":"W3155466149","doi":"10.1109/sp40001.2021.00109","title":"StochFuzz: Sound and Cost-effective Fuzzing of Stripped Binaries by Incremental and Stochastic Rewriting","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Fuzz testing; Computer science; Rewriting; Probabilistic logic; Soundness; Binary translation; Programming language; Binary number; Static analysis; Process (computing); Algorithm; Artificial intelligence; Software","score_opus":0.016390658772891135,"score_gpt":0.26720851668917894,"score_spread":0.2508178579162878,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3155466149","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053203873,0.0008852501,0.861758,0.00042185743,0.000112155634,0.00026443347,0.000527939,0.08054184,0.0022847105],"genre_scores_gemma":[0.45172077,0.000360611,0.5374214,0.0005070667,0.00007676519,0.0002332825,0.0016586271,0.0052696406,0.002751736],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995148,0.0009524032,0.00032739952,0.0007967384,0.0024147986,0.00036056357],"domain_scores_gemma":[0.9867779,0.0066320463,0.00088606874,0.0042691347,0.0012076425,0.00022723163],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030586221,0.0015003411,0.0011089261,0.0015919139,0.00075820537,0.0014200385,0.0034898394,0.0014856445,0.003044254],"category_scores_gemma":[0.018141072,0.000966015,0.001533455,0.00087222544,0.002264821,0.0027518547,0.0028843966,0.0022547352,0.0011039939],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010249261,0.00047808737,0.013062204,0.0010333767,0.00042133324,0.00071118504,0.0005688582,0.25288618,0.09542292,0.028221177,0.019756425,0.5864134],"study_design_scores_gemma":[0.00013485565,0.00029203002,0.0016004177,0.000061973726,0.00010779926,0.00046376418,0.0000501283,0.91224426,0.05731872,0.021179967,0.006466109,0.00007998334],"about_ca_topic_score_codex":0.0036821212,"about_ca_topic_score_gemma":0.0063077384,"teacher_disagreement_score":0.0036821212,"about_ca_system_score_codex":0.0010930523,"about_ca_system_score_gemma":0.002636872,"threshold_uncertainty_score":0.016175747},"labels":[],"label_agreement":null},{"id":"W3157242654","doi":"10.22215/etd/2017-11952","title":"An Analysis and Extension of Category Partition Testing in the Presence of Constraints","year":2017,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of Ottawa","funders":"","keywords":"Computer science; Constraint (computer-aided design); Partition (number theory); Set (abstract data type); Domain (mathematical analysis); Extension (predicate logic); Base (topology); Test suite; White-box testing; Completeness (order theory); Test case; Programming language; Mathematics; Software; Machine learning; Software development","score_opus":0.042013065491432446,"score_gpt":0.3265260541144748,"score_spread":0.2845129886230423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157242654","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013883641,0.00026513997,0.9751976,0.00048202625,0.00006025081,0.00010163316,0.000097150194,0.00023780575,0.009674733],"genre_scores_gemma":[0.61588794,0.0007113632,0.3729854,0.00091029616,0.000531987,0.00038781797,0.00047527897,0.0005332126,0.0075766738],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9878049,0.0037684045,0.00035930928,0.0018349474,0.0050867037,0.0011457663],"domain_scores_gemma":[0.93386495,0.05134092,0.0026957397,0.0053232647,0.0057525453,0.0010226001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007266925,0.0015655413,0.0018311135,0.003971453,0.0012319906,0.0025931208,0.004633838,0.002556107,0.005490483],"category_scores_gemma":[0.050940786,0.0009374106,0.0031350667,0.0034330545,0.0065313056,0.007927255,0.0050364123,0.0045053377,0.00063620944],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012266148,0.000099343386,0.002207655,0.00024710715,0.00008468256,0.0008331615,0.0007367997,0.12083295,0.0031645193,0.82365334,0.0022684825,0.045749307],"study_design_scores_gemma":[0.000016798358,0.00008616714,0.00077577546,0.00013282849,0.00006191843,0.00046369302,0.000080199956,0.4343874,0.0015331751,0.55848783,0.0039233486,0.00005096974],"about_ca_topic_score_codex":0.0048741708,"about_ca_topic_score_gemma":0.002374912,"teacher_disagreement_score":0.007266925,"about_ca_system_score_codex":0.0024440226,"about_ca_system_score_gemma":0.0028916192,"threshold_uncertainty_score":0.038431644},"labels":[],"label_agreement":null},{"id":"W3160700180","doi":"10.1109/icse43902.2021.00138","title":"Automatic Unit Test Generation for Machine Learning Libraries: How Far Are We?","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Unit testing; Test suite; Machine learning; Computer science; Artificial intelligence; Test (biology); Test Management Approach; Code coverage; Unit (ring theory); Test set; Set (abstract data type); Software; Quality (philosophy); Test case; Software development; Programming language; Software construction","score_opus":0.06743097283882477,"score_gpt":0.2686211446901192,"score_spread":0.20119017185129445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3160700180","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43555397,0.09744675,0.356097,0.07222625,0.0008182723,0.0006103446,0.0010823064,0.01870642,0.017458726],"genre_scores_gemma":[0.6990278,0.014313505,0.27363682,0.005359166,0.00043769798,0.00032318858,0.0020543223,0.0019867832,0.0028607058],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9865042,0.0061684884,0.000796431,0.0015204262,0.004037767,0.00097276556],"domain_scores_gemma":[0.9173309,0.05014495,0.0059798746,0.009682321,0.014016882,0.0028450813],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015514131,0.0017707928,0.0013351764,0.002758745,0.0009805649,0.004781687,0.004139474,0.0026946575,0.00375675],"category_scores_gemma":[0.076981165,0.00056214246,0.0010835759,0.0028081376,0.002082185,0.0154655855,0.0024676449,0.0033872584,0.0023930294],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048558108,0.0005593731,0.028493257,0.0011051937,0.00013834072,0.00019002809,0.00085303443,0.008532635,0.0069940314,0.0073233824,0.010869599,0.9344557],"study_design_scores_gemma":[0.0010990581,0.008654247,0.10322547,0.013997002,0.0016892374,0.0048433184,0.01711023,0.40371946,0.086084396,0.12462387,0.234102,0.000851756],"about_ca_topic_score_codex":0.0043530804,"about_ca_topic_score_gemma":0.005393631,"teacher_disagreement_score":0.015514131,"about_ca_system_score_codex":0.0014322789,"about_ca_system_score_gemma":0.003922177,"threshold_uncertainty_score":0.08204758},"labels":[],"label_agreement":null},{"id":"W3160705248","doi":"10.1145/3249769","title":"Session details: Session 48: formal specification and verification testbench generation","year":2006,"lang":"en","type":"article","venue":"Design Automation Conference","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Session (web analytics); Computer science; Programming language; Formal verification; Formal specification; Formal methods; Functional verification; Software engineering; Computer architecture; World Wide Web","score_opus":0.0665673360729413,"score_gpt":0.2736821927955757,"score_spread":0.2071148567226344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3160705248","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02694551,0.002413449,0.69989294,0.00853716,0.010373164,0.0061630406,0.016457496,0.055051465,0.1741658],"genre_scores_gemma":[0.261217,0.0018004656,0.18059449,0.002484819,0.0038101717,0.00565997,0.043483384,0.0148295425,0.48612022],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99767584,0.00077170535,0.00011555352,0.00038389157,0.0006443847,0.00040864907],"domain_scores_gemma":[0.9945695,0.0013761972,0.000114941766,0.0016640557,0.0016433799,0.000631935],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0047926577,0.0018955835,0.0020621852,0.0012458373,0.002023034,0.0031078574,0.0013615368,0.003035328,0.35702464],"category_scores_gemma":[0.006448567,0.00074457476,0.0018695768,0.0007360542,0.00053361326,0.0023579437,0.0029437363,0.0022409458,0.14605612],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026617146,0.0006797448,0.002108282,0.000512864,0.00012543018,0.00042675476,0.00042048842,0.0034713417,0.04800479,0.010376714,0.64988923,0.28132266],"study_design_scores_gemma":[0.0010544978,0.0014473337,0.005548491,0.00014484639,0.00016518266,0.00106874,0.00026719391,0.036168236,0.122869894,0.023936873,0.80708873,0.00024010502],"about_ca_topic_score_codex":0.001475756,"about_ca_topic_score_gemma":0.0015707884,"teacher_disagreement_score":0.64297533,"about_ca_system_score_codex":0.00092198607,"about_ca_system_score_gemma":0.002711003,"threshold_uncertainty_score":0.9171263},"labels":[],"label_agreement":null},{"id":"W3161421710","doi":"10.1007/978-3-030-74296-6_32","title":"Automated Repair of Layout Bugs in Web Pages with Linear Programming","year":2021,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Computer science; Solver; Integer programming; Container (type theory); Linear programming; Web page; State (computer science); Programming language; Parallel computing; Algorithm; World Wide Web; Engineering","score_opus":0.019652387797870846,"score_gpt":0.2611711895946186,"score_spread":0.24151880179674776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3161421710","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052189976,0.0003437075,0.91653204,0.00022679016,0.00005490385,0.00007900546,0.00016032996,0.026512524,0.0039008162],"genre_scores_gemma":[0.3454806,0.00020949269,0.644632,0.000107489315,0.000027263119,0.000074636824,0.00042258878,0.0019237134,0.0071222726],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987558,0.0003571003,0.00007979482,0.00025282727,0.00038457807,0.00016997832],"domain_scores_gemma":[0.9951066,0.0029645998,0.0004652284,0.00086168695,0.0005104966,0.00009139799],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008886407,0.0009640942,0.0008111054,0.0008345911,0.00065767404,0.0011816781,0.0020553372,0.00088345865,0.007711562],"category_scores_gemma":[0.005310628,0.000954091,0.0008453932,0.001055098,0.00092710886,0.0020942828,0.0016957758,0.0014518816,0.0023290287],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058759196,0.00036573145,0.0029757812,0.0006354962,0.00007965289,0.000533787,0.0004955201,0.10526855,0.045170378,0.013796619,0.014793144,0.8152977],"study_design_scores_gemma":[0.00011151592,0.00029602041,0.0011492405,0.00010049127,0.00008626203,0.00052758615,0.0002179587,0.8984508,0.06247438,0.0290997,0.007437607,0.00004848303],"about_ca_topic_score_codex":0.0031068516,"about_ca_topic_score_gemma":0.003918565,"teacher_disagreement_score":0.007711562,"about_ca_system_score_codex":0.00062766916,"about_ca_system_score_gemma":0.0010432339,"threshold_uncertainty_score":0.025797725},"labels":[],"label_agreement":null},{"id":"W3163160051","doi":"10.1145/2637365.2517225","title":"On the simplicity of synthesizing linked data structure operations","year":2013,"lang":"en","type":"article","venue":"ACM SIGPLAN Notices","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Simplicity; Data structure; Code (set theory); Simple (philosophy); Code generation; Programming language; Algorithm; Theoretical computer science; Parallel computing; Set (abstract data type); Operating system","score_opus":0.06814676140076811,"score_gpt":0.28931906523018935,"score_spread":0.22117230382942124,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3163160051","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014197795,0.000107541455,0.9775612,0.00029771234,0.00003003713,0.00007765447,0.000050937197,0.0011115663,0.0065656193],"genre_scores_gemma":[0.17758885,0.00025827644,0.8161151,0.00027327106,0.00003220158,0.0001736533,0.00019849635,0.00061456684,0.004745534],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99883753,0.00032871775,0.00009150754,0.00024063865,0.00040135675,0.000100212426],"domain_scores_gemma":[0.996375,0.002269109,0.00017772264,0.0007744291,0.00035346657,0.00005024279],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011754748,0.00058258814,0.00036218157,0.00064295233,0.0007458783,0.0017933941,0.0011681754,0.0011520002,0.0053655654],"category_scores_gemma":[0.006240336,0.0004228901,0.00087745325,0.00057246716,0.0018298973,0.0026619153,0.0014095039,0.001511874,0.0014840467],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024371623,0.00012255645,0.001521208,0.0005941185,0.00004489708,0.0004698755,0.000774097,0.1433473,0.089352265,0.5248381,0.002332549,0.23635936],"study_design_scores_gemma":[0.00013087993,0.0003576876,0.00046149245,0.00020234605,0.00008866985,0.0005494844,0.00021309184,0.45578352,0.15023133,0.34768292,0.044236396,0.00006215062],"about_ca_topic_score_codex":0.0010212968,"about_ca_topic_score_gemma":0.0021297522,"teacher_disagreement_score":0.0053655654,"about_ca_system_score_codex":0.00072755106,"about_ca_system_score_gemma":0.0008233987,"threshold_uncertainty_score":0.01794964},"labels":[],"label_agreement":null},{"id":"W3163537710","doi":"10.1145/3249746","title":"Session details: Session 25: the test bin","year":2006,"lang":"en","type":"article","venue":"Design Automation Conference","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Session (web analytics); Computer science; Test (biology); Bin; World Wide Web; Programming language","score_opus":0.04490146855196091,"score_gpt":0.27352533837223153,"score_spread":0.2286238698202706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3163537710","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0145311225,0.0026295655,0.080917776,0.016055826,0.035497934,0.0055028927,0.027423037,0.055856302,0.7615856],"genre_scores_gemma":[0.054876845,0.0006826538,0.011036414,0.0029137908,0.00475081,0.0016483516,0.01619372,0.008177995,0.8997194],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988387,0.0002337833,0.00004170029,0.00022695625,0.00033793913,0.00032094817],"domain_scores_gemma":[0.9953988,0.00062598754,0.000120125835,0.0008927143,0.001048827,0.00191355],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.002749856,0.0017556394,0.0024111678,0.0011079605,0.0028050747,0.0057202387,0.0018560181,0.003694757,0.7179901],"category_scores_gemma":[0.003907924,0.0007375583,0.0015207893,0.0009233091,0.00047338454,0.0023167105,0.004337367,0.0030064688,0.5109599],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00089518353,0.00020152803,0.00054010656,0.00010241638,0.000015672127,0.00007620846,0.000053203905,0.00018325448,0.0021303992,0.0012110528,0.9370085,0.057582546],"study_design_scores_gemma":[0.00023990846,0.00035161604,0.0036055665,0.00007429034,0.000033531363,0.0001475553,0.00013280072,0.0018063232,0.0056247967,0.0032953192,0.98462766,0.00006069887],"about_ca_topic_score_codex":0.0018080479,"about_ca_topic_score_gemma":0.003672352,"teacher_disagreement_score":0.2820099,"about_ca_system_score_codex":0.00094030617,"about_ca_system_score_gemma":0.0017413269,"threshold_uncertainty_score":0.40225285},"labels":[],"label_agreement":null},{"id":"W3163597827","doi":"10.1109/icse43902.2021.00019","title":"Studying Test Annotation Maintenance in the Wild","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Annotation; Computer science; Java; Fixture; Test (biology); Empirical research; Test case; Software engineering; Artificial intelligence; Machine learning; Programming language; Engineering","score_opus":0.031171318486716636,"score_gpt":0.2705429677314099,"score_spread":0.23937164924469326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3163597827","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9588245,0.0019976613,0.03378554,0.0015679239,0.00010351161,0.000113536604,0.0008530762,0.0010594677,0.0016946826],"genre_scores_gemma":[0.9491662,0.00069387356,0.045338567,0.00038724436,0.00006596197,0.0001587436,0.00267852,0.00055518304,0.00095578743],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9768859,0.0077979364,0.0020627684,0.0051746545,0.006877748,0.0012010538],"domain_scores_gemma":[0.686715,0.21542464,0.040455814,0.033696145,0.021032277,0.0026761533],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020267468,0.0005844491,0.0007529025,0.007752096,0.001766833,0.002992032,0.0029789573,0.0018720725,0.00072153355],"category_scores_gemma":[0.17730217,0.00085025,0.00053491537,0.0075625503,0.0024238937,0.007044217,0.0020940169,0.0021068037,0.0003058454],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039486695,0.00068425044,0.6377293,0.00082918484,0.00019215944,0.0019114587,0.018276507,0.0052478462,0.012585352,0.006349595,0.0068421974,0.3089573],"study_design_scores_gemma":[0.000120002805,0.0009957324,0.77464354,0.0012950753,0.0003593092,0.0058307406,0.02124136,0.10172737,0.026470313,0.016765246,0.05024765,0.00030373293],"about_ca_topic_score_codex":0.008405355,"about_ca_topic_score_gemma":0.0125048645,"teacher_disagreement_score":0.020267468,"about_ca_system_score_codex":0.0021790846,"about_ca_system_score_gemma":0.0020667503,"threshold_uncertainty_score":0.1071859},"labels":[],"label_agreement":null},{"id":"W3169794909","doi":"10.1109/icstw52544.2021.00040","title":"Test Sequence Generation with Cayley Graphs","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Test suite; Computer science; Sequence (biology); Graph; Theoretical computer science; Set (abstract data type); Metric (unit); Finite-state machine; Algorithm; Test case; Programming language; Engineering; Machine learning","score_opus":0.04702509873227484,"score_gpt":0.26444477978923137,"score_spread":0.21741968105695653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3169794909","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011088779,0.00006457001,0.9857817,0.0000804339,0.000012935671,0.000115836076,0.000107298845,0.0012984351,0.0014499984],"genre_scores_gemma":[0.22441089,0.00016650908,0.7723657,0.00018102335,0.000026920476,0.00030399766,0.0009417288,0.00041502708,0.0011882053],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99750876,0.00084147975,0.00017895784,0.000472323,0.0008291223,0.00016927462],"domain_scores_gemma":[0.99476767,0.003364565,0.0003985009,0.0007636507,0.00060177327,0.00010388709],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014472738,0.000748779,0.00052991166,0.0014892494,0.00041797667,0.0010238069,0.0015461118,0.00064614264,0.0029994068],"category_scores_gemma":[0.006564616,0.00048178266,0.0010394917,0.0012267289,0.0011374658,0.001587653,0.0009232875,0.0012883726,0.00059962954],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025625387,0.00016099996,0.0023810312,0.00043795595,0.00008315308,0.00061693694,0.00027869598,0.49819374,0.03726623,0.19139934,0.0040581636,0.2648675],"study_design_scores_gemma":[0.000052907588,0.00016554109,0.00026046298,0.000045161305,0.000024119357,0.0002511213,0.0000395838,0.892869,0.021499924,0.080006465,0.004759865,0.000025834475],"about_ca_topic_score_codex":0.0027065703,"about_ca_topic_score_gemma":0.0029052927,"teacher_disagreement_score":0.0029994068,"about_ca_system_score_codex":0.001235928,"about_ca_system_score_gemma":0.0013966145,"threshold_uncertainty_score":0.010033965},"labels":[],"label_agreement":null},{"id":"W3173275515","doi":"","title":"RIFF: Reduced Instruction Footprint for Coverage-Guided Fuzzing.","year":2021,"lang":"en","type":"article","venue":"USENIX Annual Technical Conference","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Fuzz testing; Computer science; Footprint; Operating system; Software; Geography","score_opus":0.0541948968657504,"score_gpt":0.3114097351044024,"score_spread":0.257214838238652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173275515","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12679407,0.0018762808,0.7398241,0.00070725934,0.00033788406,0.00025518073,0.0019266029,0.11898885,0.0092898095],"genre_scores_gemma":[0.60929304,0.00033733514,0.3742359,0.0005411703,0.000103415354,0.00031125528,0.0033144157,0.0048292167,0.007034233],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983784,0.00034335768,0.000089235386,0.00025375138,0.0007310752,0.00020406694],"domain_scores_gemma":[0.9953275,0.0021353855,0.00026255514,0.0015483508,0.0005816977,0.00014450197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011329795,0.0013679406,0.0009700897,0.0016675727,0.0005549306,0.00088621303,0.0030029323,0.0014149871,0.008499946],"category_scores_gemma":[0.0066452855,0.00063880073,0.00073691364,0.0008408717,0.00082289544,0.0025979213,0.0015478521,0.0015650443,0.0019416728],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002665458,0.00060503883,0.0052034515,0.0007540033,0.0002450691,0.00076213497,0.00041048517,0.07504595,0.14735614,0.02033323,0.054980718,0.69163835],"study_design_scores_gemma":[0.00040195312,0.0008624988,0.002384199,0.00012855124,0.00014867318,0.00063214824,0.00011677262,0.78288066,0.17019895,0.024321219,0.017799668,0.0001247489],"about_ca_topic_score_codex":0.003314585,"about_ca_topic_score_gemma":0.005869608,"teacher_disagreement_score":0.008499946,"about_ca_system_score_codex":0.00066584477,"about_ca_system_score_gemma":0.0014036021,"threshold_uncertainty_score":0.02843517},"labels":[],"label_agreement":null},{"id":"W3173309977","doi":"10.1109/msr52588.2021.00084","title":"EqBench: A Dataset of Equivalent and Non-equivalent Program Pairs","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Ministry of Education","keywords":"Equivalence (formal languages); Computer science; Formal equivalence checking; Programming language; Java; Benchmark (surveying); Model checking; Program analysis; String (physics); Theoretical computer science; Algorithm; Arithmetic; Discrete mathematics; Mathematics","score_opus":0.033712847771385834,"score_gpt":0.3087434225400677,"score_spread":0.27503057476868187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173309977","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13683751,0.005911137,0.03842921,0.0012078382,0.00043157788,0.00069326523,0.71326756,0.07441054,0.028811308],"genre_scores_gemma":[0.08586201,0.00094544044,0.0328871,0.00055942783,0.000049210663,0.00076357456,0.8713085,0.005015068,0.0026095735],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99259406,0.0012193319,0.001128144,0.0015955943,0.0028863852,0.00057652517],"domain_scores_gemma":[0.98301506,0.009108555,0.0012658995,0.0031509004,0.002891142,0.00056843506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028795856,0.0026557345,0.0008730241,0.0064948136,0.0011052797,0.0017698624,0.0037435472,0.002239962,0.00894305],"category_scores_gemma":[0.025815174,0.0007518908,0.0015807616,0.0050084805,0.0010758343,0.0031471457,0.0032600872,0.0020544922,0.004414822],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023787292,0.000995331,0.05558567,0.011777238,0.000834901,0.0026171461,0.0008551061,0.028933669,0.015761439,0.021072434,0.70014834,0.15904006],"study_design_scores_gemma":[0.0013367649,0.00088524725,0.07096143,0.0015963485,0.0004824775,0.0029375292,0.0008955165,0.0797239,0.03549318,0.038164996,0.7671768,0.0003458228],"about_ca_topic_score_codex":0.0069817407,"about_ca_topic_score_gemma":0.012493087,"teacher_disagreement_score":0.00894305,"about_ca_system_score_codex":0.0014834222,"about_ca_system_score_gemma":0.0021347713,"threshold_uncertainty_score":0.029917479},"labels":[],"label_agreement":null},{"id":"W3174050238","doi":"10.1109/icccs52626.2021.9449249","title":"Automatic Generation of Parallel Java Programs and their Validation using Combinatorial Testing Suites","year":2021,"lang":"en","type":"article","venue":"2021 IEEE 6th International Conference on Computer and Communication Systems (ICCCS)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Stem Cell Network","keywords":"Computer science; Executable; Java; Correctness; Programming language; Test suite; Parallel computing; Thread (computing); Suite; Multi-core processor; Operating system; Test case","score_opus":0.179823003380709,"score_gpt":0.3190231130519308,"score_spread":0.13920010967122182,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3174050238","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12596709,0.000082374965,0.8585991,0.0000919572,0.000031035954,0.00024704184,0.00021642685,0.012681994,0.0020828967],"genre_scores_gemma":[0.55349874,0.00011570797,0.44159174,0.000093637864,0.000021444255,0.00047238887,0.0013092782,0.002016283,0.00088076695],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99507856,0.0018011257,0.0004070614,0.0007969138,0.0015780549,0.00033822388],"domain_scores_gemma":[0.9895375,0.006583459,0.0009708792,0.0015338508,0.0012101937,0.00016422657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002601565,0.0011237843,0.0006966123,0.0015842693,0.0004426318,0.0013367884,0.0015808782,0.0006626586,0.0018639092],"category_scores_gemma":[0.013660826,0.0005273213,0.0013489871,0.0008315702,0.0013620373,0.0014036401,0.0013202783,0.0008436804,0.00046343135],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010156961,0.0009648044,0.018966196,0.0010001017,0.00027932253,0.002574952,0.00090744137,0.3057189,0.2445958,0.049738564,0.0031001698,0.3711381],"study_design_scores_gemma":[0.00013748342,0.00044130464,0.0025853517,0.0000829951,0.00008262779,0.0008040323,0.00009533757,0.8274152,0.14005895,0.024184892,0.004049956,0.000061770144],"about_ca_topic_score_codex":0.0008556642,"about_ca_topic_score_gemma":0.000652499,"teacher_disagreement_score":0.002601565,"about_ca_system_score_codex":0.00062938087,"about_ca_system_score_gemma":0.0009165866,"threshold_uncertainty_score":0.01375854},"labels":[],"label_agreement":null},{"id":"W3178329914","doi":"10.1145/3460319.3464824","title":"Log-based slicing for system-level test cases","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg","keywords":"Computer science; Slicing; Regression testing; Program slicing; Test (biology); Test suite; Test Management Approach; Test case; Test harness; System under test; Code coverage; Reliability engineering; Test script; White-box testing; Software; Programming language; Software system; Machine learning; Regression analysis; Engineering; Software construction","score_opus":0.07594283484011646,"score_gpt":0.2931907843325324,"score_spread":0.2172479494924159,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3178329914","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016730784,0.00021547714,0.9620886,0.00012412886,0.000039961953,0.00033586015,0.0007102237,0.016917946,0.002837019],"genre_scores_gemma":[0.36153114,0.00037924876,0.6288934,0.00017489493,0.000055382996,0.0006858472,0.0032145793,0.0029312472,0.0021343478],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99702615,0.00088126754,0.000342572,0.0004110617,0.0010787799,0.0002602509],"domain_scores_gemma":[0.9821832,0.009912647,0.0017917852,0.0037994673,0.0020149034,0.00029795352],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028689983,0.0013819989,0.0006968679,0.0025243894,0.0004248929,0.0013292585,0.0013459943,0.00052900077,0.0072225435],"category_scores_gemma":[0.018100217,0.0007467812,0.0010119409,0.001071176,0.0013214641,0.00269677,0.0013683315,0.0014861731,0.0012775509],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007508985,0.00039281545,0.011349632,0.0010738111,0.00012247558,0.001256325,0.00083084905,0.25055912,0.04727512,0.06086038,0.010853706,0.61467487],"study_design_scores_gemma":[0.000094659255,0.00027323695,0.0023534682,0.00025154153,0.000082525796,0.0004831783,0.00012866071,0.873537,0.05337786,0.053988397,0.015358005,0.00007154419],"about_ca_topic_score_codex":0.0042037787,"about_ca_topic_score_gemma":0.0055246074,"teacher_disagreement_score":0.0072225435,"about_ca_system_score_codex":0.0009728701,"about_ca_system_score_gemma":0.0019680387,"threshold_uncertainty_score":0.024161756},"labels":[],"label_agreement":null},{"id":"W3195549714","doi":"10.1145/3468264.3473123","title":"Slicer4J: a dynamic slicer for Java","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Java; Program slicing; Call graph; Debugging; Programming language; Slicing; Java annotation; Control flow graph; Real time Java; Multithreading; Thread (computing); Java concurrency; Speculative multithreading; Static analysis; Scala; Control flow; strictfp; Data-flow analysis; TRACE (psycholinguistics); Data flow diagram; Database","score_opus":0.016872314729178182,"score_gpt":0.2906664266073143,"score_spread":0.2737941118781361,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3195549714","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0076030428,0.0007415591,0.67331874,0.00020663427,0.00022421619,0.0002551661,0.0027237502,0.30929968,0.005627269],"genre_scores_gemma":[0.11562506,0.0011927547,0.7388967,0.00066959014,0.00014121973,0.00065576984,0.016927285,0.116409086,0.009482492],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99697065,0.00044975328,0.0003568416,0.0005815854,0.0013368774,0.00030418718],"domain_scores_gemma":[0.9953868,0.0015699861,0.0004988761,0.0013469346,0.0009945085,0.00020301045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033459952,0.002027011,0.00097259355,0.0024420714,0.0007089434,0.0021326502,0.0033823315,0.0012761987,0.008482371],"category_scores_gemma":[0.007428278,0.0017969871,0.002460507,0.0011795093,0.001442484,0.004215396,0.0026912063,0.0030325556,0.003375982],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022417328,0.00041279066,0.012365041,0.003091354,0.0006192072,0.0011189177,0.001965432,0.025257463,0.10949683,0.056910064,0.1911872,0.595334],"study_design_scores_gemma":[0.0007279535,0.00070334406,0.0068972413,0.00067982596,0.00039747945,0.0019275837,0.0002921657,0.24098817,0.2153449,0.047737923,0.48363152,0.0006719138],"about_ca_topic_score_codex":0.004813239,"about_ca_topic_score_gemma":0.0063145226,"teacher_disagreement_score":0.008482371,"about_ca_system_score_codex":0.00088826695,"about_ca_system_score_gemma":0.0027320907,"threshold_uncertainty_score":0.02837634},"labels":[],"label_agreement":null},{"id":"W3195614596","doi":"10.1145/3468264.3473930","title":"Quantifying no-fault-found test failures to prioritize inspection of flaky tests at Ericsson","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Reliability engineering; Fault (geology); Test (biology); Fault coverage; Work (physics); Software; Computer science; Software testing; Engineering; Embedded system; Operating system; Electrical engineering; Geology; Mechanical engineering; Seismology","score_opus":0.039810804368580484,"score_gpt":0.308410739023143,"score_spread":0.2685999346545625,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3195614596","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.978157,0.00080027344,0.018055923,0.00020739557,0.000029995792,0.00006067504,0.00072092685,0.0008097748,0.0011580156],"genre_scores_gemma":[0.98136294,0.00012196332,0.016719466,0.000055295022,0.00001321986,0.000050122195,0.0010303383,0.00016107698,0.00048549182],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9890342,0.0034504994,0.0010131601,0.0021479167,0.0034886452,0.0008656254],"domain_scores_gemma":[0.8529387,0.097912565,0.022790436,0.006959769,0.016362432,0.003035972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009581443,0.0010958455,0.0007799289,0.0063774157,0.00054781645,0.0018082398,0.0013899627,0.00143221,0.0010077058],"category_scores_gemma":[0.08137291,0.0004704265,0.0005369549,0.003043244,0.00086852454,0.0023275875,0.0013422468,0.0011898715,0.00045776714],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011015408,0.00057671074,0.79390633,0.0006762834,0.000480063,0.0007421352,0.0020109038,0.06533787,0.019140473,0.0013839336,0.0033222088,0.1113215],"study_design_scores_gemma":[0.000120086166,0.0017822799,0.67059577,0.00030569927,0.00035974607,0.0013572207,0.0024431478,0.28427437,0.02969243,0.0036224462,0.0052734935,0.00017331635],"about_ca_topic_score_codex":0.009318043,"about_ca_topic_score_gemma":0.01543841,"teacher_disagreement_score":0.009581443,"about_ca_system_score_codex":0.0011240548,"about_ca_system_score_gemma":0.0012009798,"threshold_uncertainty_score":0.050672114},"labels":[],"label_agreement":null},{"id":"W3196239222","doi":"10.1145/3468264.3468591","title":"A comprehensive study of deep learning compiler bugs","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":112,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Compiler; Computer science; Programming language; Boosting (machine learning); Deep learning; Context (archaeology); Code (set theory); Code generation; Parallel computing; Artificial intelligence; Operating system","score_opus":0.03173090442745961,"score_gpt":0.28398372578950304,"score_spread":0.2522528213620434,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196239222","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25165418,0.034928694,0.67499673,0.013959049,0.00059032656,0.00010806039,0.0005866203,0.0030762278,0.020100173],"genre_scores_gemma":[0.8845194,0.008845049,0.09753366,0.0013861606,0.0004888463,0.00009539775,0.00049586734,0.0007039179,0.005931674],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.9977367,0.00046663333,0.00013967635,0.0003975914,0.0010219979,0.00023734928],"domain_scores_gemma":[0.9842939,0.010286476,0.0016908637,0.0013618845,0.0020743802,0.00029249923],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003163037,0.0009864264,0.0009047389,0.0018238183,0.0006628768,0.0015526903,0.001748697,0.0015015091,0.0016919981],"category_scores_gemma":[0.027564248,0.00091371086,0.00077325373,0.0014205335,0.0026714113,0.0042367987,0.0015358361,0.003144266,0.00028567016],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020065492,0.00030422374,0.021582682,0.0016790138,0.00027131295,0.0006008615,0.0006300489,0.33450404,0.005439697,0.28860995,0.016190866,0.3299867],"study_design_scores_gemma":[0.000025511328,0.00012345471,0.003662958,0.00039444194,0.00007936738,0.00042501447,0.00011128766,0.7019027,0.0056161364,0.27694845,0.010648842,0.00006191315],"about_ca_topic_score_codex":0.0047401176,"about_ca_topic_score_gemma":0.003711327,"teacher_disagreement_score":0.99683696,"about_ca_system_score_codex":0.0019516095,"about_ca_system_score_gemma":0.0023216887,"threshold_uncertainty_score":0.016727924},"labels":[],"label_agreement":null},{"id":"W3196351123","doi":"","title":"Leveraging Documentation to Test Deep Learning Library Functions","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Fuzz testing; Constraint (computer-aided design); Code (set theory); Artificial intelligence; Software bug; Machine learning; Data mining; Software; Programming language; Set (abstract data type)","score_opus":0.05306870260441219,"score_gpt":0.19408058708966935,"score_spread":0.14101188448525717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196351123","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53527457,0.0018914585,0.29372522,0.0023283835,0.00029376685,0.0005695552,0.0057070274,0.14813417,0.01207585],"genre_scores_gemma":[0.7930176,0.00046866562,0.18752876,0.0010629223,0.000047386577,0.00029118877,0.007083139,0.007491506,0.0030089074],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9862147,0.0027555402,0.0015811411,0.0021997206,0.0065893675,0.00065961573],"domain_scores_gemma":[0.9151898,0.04443846,0.007830482,0.017771268,0.013649082,0.0011209124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007977188,0.0016993913,0.0006466515,0.002973624,0.0007161096,0.002559815,0.00311554,0.0015188738,0.0029411959],"category_scores_gemma":[0.079042986,0.0012585326,0.0009961488,0.0015517346,0.0016127656,0.006340321,0.0024950386,0.0024223803,0.0016732129],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013768302,0.0010928517,0.12556648,0.0020240785,0.00036744,0.0018715453,0.0015024992,0.06755872,0.050829757,0.017758476,0.044400487,0.6856509],"study_design_scores_gemma":[0.0002885362,0.0007722925,0.020296315,0.0006795541,0.00020208722,0.0009910476,0.00038175762,0.6530114,0.25770932,0.02344858,0.041985393,0.00023380671],"about_ca_topic_score_codex":0.0074519343,"about_ca_topic_score_gemma":0.0095394375,"teacher_disagreement_score":0.007977188,"about_ca_system_score_codex":0.0022642121,"about_ca_system_score_gemma":0.003818718,"threshold_uncertainty_score":0.04218793},"labels":[],"label_agreement":null},{"id":"W3196600806","doi":"","title":"Finding Counterexamples of Temporal Logic properties in Software Implementations via Greybox Fuzzing","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Fuzz testing; Computer science; Model checking; Liveness; Programming language; Counterexample; Temporal logic; Stateful firewall; Software; Property (philosophy); Linear temporal logic; Implementation; Theoretical computer science; Mathematics; Computer security","score_opus":0.1810027213860887,"score_gpt":0.23743890492983932,"score_spread":0.05643618354375063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196600806","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41182405,0.000258564,0.5797044,0.0004823616,0.000058744565,0.00015232865,0.0002799636,0.0058522522,0.0013873101],"genre_scores_gemma":[0.8169208,0.00011583708,0.18136497,0.00023021692,0.0000163896,0.00012172547,0.00035370135,0.00030473492,0.00057153194],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964168,0.0006745031,0.00023029614,0.0009181696,0.0014814367,0.00027870142],"domain_scores_gemma":[0.9819838,0.013070901,0.0017054636,0.0020385212,0.0009710954,0.00023017809],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022160113,0.0008035195,0.0006612206,0.001995152,0.00056193705,0.001331892,0.001209115,0.0013817475,0.0014673119],"category_scores_gemma":[0.023659095,0.0006393797,0.0016807242,0.0008150191,0.0018172953,0.0025534662,0.0016015886,0.0014675648,0.00018532043],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013666438,0.0004950504,0.064704984,0.0010082624,0.00058442465,0.0046453904,0.0024797148,0.41320944,0.18842782,0.12505443,0.0034707857,0.194553],"study_design_scores_gemma":[0.00008373062,0.00016707704,0.0030400683,0.00011250374,0.00013566633,0.00055213645,0.00014156675,0.88155854,0.062817745,0.049586963,0.0017415392,0.00006250434],"about_ca_topic_score_codex":0.0031821856,"about_ca_topic_score_gemma":0.0036048598,"teacher_disagreement_score":0.0031821856,"about_ca_system_score_codex":0.0011509061,"about_ca_system_score_gemma":0.0013429839,"threshold_uncertainty_score":0.011719525},"labels":[],"label_agreement":null},{"id":"W3197204261","doi":"10.5281/zenodo.5346846","title":"Mapping breakpoint types: an exploratory study","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Breakpoint; Computer science; Cartography; Computational biology; Geography; Biology; Genetics; Chromosomal translocation","score_opus":0.0637448935379894,"score_gpt":0.27887108849297804,"score_spread":0.21512619495498864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3197204261","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008072435,0.0006493408,0.0008313521,0.00018345314,0.00003546799,0.00012198898,0.98719066,0.00042198866,0.0024933831],"genre_scores_gemma":[0.0056284545,0.00029926465,0.0019785895,0.00010964862,0.000011684581,0.00048700103,0.99013114,0.00011602462,0.0012382723],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99681133,0.00072303717,0.000712784,0.00070664345,0.0007399358,0.00030632288],"domain_scores_gemma":[0.9874986,0.0058223642,0.0014652469,0.0017943064,0.0029118913,0.00050759007],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026151538,0.00097344036,0.0008359708,0.010494972,0.0010339781,0.0020898937,0.0016518609,0.0015601388,0.013367985],"category_scores_gemma":[0.015267735,0.00040684003,0.0008300715,0.012926963,0.00044627668,0.0014070221,0.0022869697,0.0012551814,0.016354527],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072210975,0.00029969247,0.03765771,0.008113809,0.00017396572,0.0006086215,0.0010836327,0.00086516753,0.0019679323,0.003633249,0.90502197,0.039852094],"study_design_scores_gemma":[0.0002647583,0.00007320254,0.04228668,0.0013353359,0.00008826224,0.0004846273,0.0012668154,0.0009104492,0.0019497523,0.0015731468,0.9497085,0.00005842218],"about_ca_topic_score_codex":0.01106621,"about_ca_topic_score_gemma":0.024670826,"teacher_disagreement_score":0.013367985,"about_ca_system_score_codex":0.001635831,"about_ca_system_score_gemma":0.0021194397,"threshold_uncertainty_score":0.04472035},"labels":[],"label_agreement":null},{"id":"W3197206692","doi":"10.1109/tse.2021.3107680","title":"Mutation Analysis for Cyber-Physical Systems: Scalable Solutions and Results in the Space Domain","year":2022,"lang":"en","type":"article","venue":"Open Repository and Bibliography (University of Luxembourg)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"H2020 European Research Council; Natural Sciences and Engineering Research Council of Canada; European Commission; European Space Agency","keywords":"Computer science; Scalability; Cyber-physical system; Domain (mathematical analysis); Space (punctuation); Domain analysis; Theoretical computer science; Distributed computing; Computer security; Programming language; Operating system; Software; Software development","score_opus":0.024208405854101948,"score_gpt":0.24324409481134837,"score_spread":0.21903568895724643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3197206692","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08338792,0.0018708017,0.89795756,0.001254424,0.00020806181,0.00015546194,0.00035480745,0.0053677456,0.009443192],"genre_scores_gemma":[0.6494173,0.0012053082,0.3377313,0.0003262824,0.00017067067,0.0001787805,0.00077515235,0.0011644542,0.009030774],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981092,0.0005682264,0.000071184906,0.0003280007,0.0007537834,0.00016962674],"domain_scores_gemma":[0.9899254,0.007006382,0.000329819,0.0012290803,0.0012430192,0.00026638],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018909196,0.0012774294,0.0017428254,0.0023782118,0.00076026877,0.0014883738,0.0015330497,0.0014143472,0.006939309],"category_scores_gemma":[0.012411722,0.00029098446,0.0015626283,0.001891879,0.0015068796,0.003089379,0.0015824257,0.0017288439,0.0008912023],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060574675,0.00055401825,0.005301813,0.0011565803,0.00032796347,0.00046406584,0.00030713258,0.36734492,0.023571726,0.06033976,0.014211588,0.5258147],"study_design_scores_gemma":[0.00012124947,0.00014313572,0.0016931085,0.00008525833,0.00010538647,0.00013820961,0.000103328995,0.90315145,0.010254938,0.08158108,0.0025906188,0.000032171112],"about_ca_topic_score_codex":0.0030948617,"about_ca_topic_score_gemma":0.003467275,"teacher_disagreement_score":0.006939309,"about_ca_system_score_codex":0.0009063921,"about_ca_system_score_gemma":0.0016870204,"threshold_uncertainty_score":0.02321434},"labels":[],"label_agreement":null},{"id":"W3201392238","doi":"10.1007/s10270-021-00918-6","title":"Automated generation of consistent models using qualitative abstractions and exploration strategies","year":2021,"lang":"en","type":"article","venue":"Software & Systems Modeling","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Nemzeti Kutatási, Fejlesztési és Innovaciós Alap; Natural Sciences and Engineering Research Council of Canada; National Research, Development and Innovation Office; Innovációs és Technológiai Minisztérium; Nemzeti Kutatási Fejlesztési és Innovációs Hivatal","keywords":"Computer science; Solver; Scalability; Key (lock); Context (archaeology); Graph; Consistency (knowledge bases); Theoretical computer science; Programming language; Artificial intelligence","score_opus":0.2661288694789344,"score_gpt":0.37461296274791683,"score_spread":0.10848409326898245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3201392238","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01412063,0.000033118034,0.983499,0.00011551506,0.0000072481002,0.00008614604,0.00008687229,0.0008019856,0.0012494964],"genre_scores_gemma":[0.2333946,0.00009541644,0.7645535,0.000057250927,0.0000057592943,0.00027439868,0.0003457065,0.0003257468,0.0009476307],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984037,0.00058792415,0.000086672844,0.00018953055,0.0006150643,0.00011712215],"domain_scores_gemma":[0.99529725,0.0032696337,0.00028338938,0.0006573567,0.00041803467,0.00007440871],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025777258,0.0010921916,0.0005228982,0.0011881125,0.0005133286,0.0013841465,0.0017290153,0.0008822154,0.0034025623],"category_scores_gemma":[0.010669586,0.0007344766,0.0014893507,0.00058370916,0.0015371813,0.0019266952,0.0026745447,0.0011994519,0.00044906323],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008764046,0.00009995194,0.0019772225,0.00031938628,0.000055288645,0.00031025996,0.00056440633,0.8090784,0.014831157,0.08965692,0.0010576866,0.08196174],"study_design_scores_gemma":[0.000023721472,0.000040831164,0.00008994177,0.000030429095,0.000014902291,0.00005561253,0.0001028262,0.94601923,0.005644067,0.04587786,0.002089834,0.000010735263],"about_ca_topic_score_codex":0.0019908377,"about_ca_topic_score_gemma":0.0031579246,"teacher_disagreement_score":0.0034025623,"about_ca_system_score_codex":0.0010272034,"about_ca_system_score_gemma":0.0019895998,"threshold_uncertainty_score":0.013632476},"labels":[],"label_agreement":null},{"id":"W3207636857","doi":"10.1002/stvr.1796","title":"GPU acceleration of finite state machine input execution: Improving scale and performance","year":2021,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Engineering and Physical Sciences Research Council; University of Edinburgh","keywords":"Computer science; Scalability; Parallel computing; Kernel (algebra); Speedup; Multi-core processor; Finite-state machine; Acceleration; Process (computing); CUDA; Computer engineering; Algorithm; Programming language; Operating system","score_opus":0.02022802077216418,"score_gpt":0.24044934009781616,"score_spread":0.22022131932565198,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3207636857","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.67771345,0.0029282297,0.23497503,0.00082568213,0.0006153285,0.0002513567,0.0010310018,0.055176537,0.02648347],"genre_scores_gemma":[0.85453975,0.00031806002,0.13994682,0.00013407612,0.000030772313,0.000108715125,0.0013245785,0.0011964111,0.0024008465],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99875724,0.00024306963,0.00007334766,0.00023171733,0.0005072189,0.00018745742],"domain_scores_gemma":[0.99684864,0.0013139572,0.00013133226,0.0008716639,0.00069740094,0.00013699493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010774153,0.00102729,0.00068593374,0.00084503525,0.0003760081,0.0010672203,0.0020596255,0.000504174,0.004973635],"category_scores_gemma":[0.0054893843,0.00043343086,0.0007423137,0.0010173973,0.00038692376,0.0013636592,0.0008476615,0.00111137,0.0011970497],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031042078,0.00092328625,0.024845457,0.0009733742,0.00037813364,0.00048697423,0.00065380044,0.24930945,0.117883526,0.011628344,0.036509357,0.5533042],"study_design_scores_gemma":[0.00013156854,0.00025615236,0.0027711536,0.000036918333,0.00005388137,0.000057297595,0.00006593296,0.94789326,0.039926264,0.002085867,0.006690685,0.000030939136],"about_ca_topic_score_codex":0.008710368,"about_ca_topic_score_gemma":0.009333426,"teacher_disagreement_score":0.008710368,"about_ca_system_score_codex":0.0011081463,"about_ca_system_score_gemma":0.0011303115,"threshold_uncertainty_score":0.017319322},"labels":[],"label_agreement":null},{"id":"W3210529140","doi":"10.1109/tse.2022.3171295","title":"Fragment-Based Test Generation for Web Apps","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Test (biology); Fragment (logic); Web application; Software engineering; World Wide Web; Programming language; Operating system; Database","score_opus":0.02069671311464428,"score_gpt":0.22918178894011498,"score_spread":0.2084850758254707,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3210529140","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09182589,0.00019487993,0.85309553,0.00033366177,0.00006304708,0.0005043897,0.0011522243,0.049281936,0.0035484405],"genre_scores_gemma":[0.6738239,0.00013226755,0.31693915,0.00030162334,0.000029189663,0.0005612797,0.0033628964,0.0026226728,0.0022271343],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99674964,0.0010663024,0.00021343853,0.0004718731,0.0012493642,0.00024939125],"domain_scores_gemma":[0.9904704,0.0055458914,0.00057092553,0.0021016144,0.0011485941,0.00016261076],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016384395,0.0012098743,0.00056859513,0.0014871733,0.0003385599,0.0008740958,0.0023680767,0.0010646315,0.0047206716],"category_scores_gemma":[0.014549616,0.0005574336,0.0011243242,0.0007223791,0.0008969534,0.0018954085,0.0014657206,0.0011239557,0.00092873466],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013874973,0.0006958538,0.0116729075,0.0005454867,0.00016974116,0.0012446387,0.00051960786,0.4353633,0.06595454,0.021728596,0.019302122,0.4414158],"study_design_scores_gemma":[0.00009465532,0.0002910358,0.0008719799,0.0000310649,0.000033482127,0.00022962329,0.00003199759,0.9532752,0.031603504,0.010059671,0.0034465198,0.000031245134],"about_ca_topic_score_codex":0.0042573754,"about_ca_topic_score_gemma":0.003842487,"teacher_disagreement_score":0.0047206716,"about_ca_system_score_codex":0.0010411941,"about_ca_system_score_gemma":0.001331866,"threshold_uncertainty_score":0.015792191},"labels":[],"label_agreement":null},{"id":"W3212114119","doi":"10.1109/models50736.2021.00018","title":"Efficient Replay-based Regression Testing for Distributed Reactive Systems in the Context of Model-driven Development","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Regression testing; Computer science; Context (archaeology); Overhead (engineering); Timestamp; Distributed computing; Regression; Software; Data mining; Machine learning; Software system; Real-time computing; Programming language","score_opus":0.06536512290365601,"score_gpt":0.2960268156135097,"score_spread":0.23066169270985368,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3212114119","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20195808,0.0005243877,0.7797035,0.00032418562,0.00003649106,0.00011359231,0.00008601466,0.016407203,0.00084659446],"genre_scores_gemma":[0.81446695,0.00011715131,0.18425885,0.000062856496,0.00001055094,0.000068110654,0.00015486524,0.00040714876,0.00045354562],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9959461,0.0020299796,0.00017438468,0.0006177017,0.00097060297,0.00026123328],"domain_scores_gemma":[0.988865,0.006744865,0.0011375032,0.0022353427,0.000780112,0.00023712813],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002779636,0.0008527815,0.0007516913,0.00080138835,0.00036022105,0.0008753576,0.0016711473,0.0007495068,0.0007830137],"category_scores_gemma":[0.014231634,0.0005251056,0.00048193286,0.000432775,0.0008129265,0.0016039533,0.0011819921,0.0012571746,0.00023535157],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012879011,0.0007423097,0.014517751,0.0005271624,0.00015813025,0.0010432458,0.0008967441,0.53143895,0.10940784,0.010046399,0.002784967,0.32714862],"study_design_scores_gemma":[0.00003529779,0.00020193381,0.00091931986,0.000014377057,0.000026021893,0.00015789845,0.00004066883,0.97181904,0.02286606,0.0030971824,0.00080289674,0.000019360416],"about_ca_topic_score_codex":0.0037251527,"about_ca_topic_score_gemma":0.003717,"teacher_disagreement_score":0.0037251527,"about_ca_system_score_codex":0.0006730462,"about_ca_system_score_gemma":0.0012647682,"threshold_uncertainty_score":0.014700294},"labels":[],"label_agreement":null},{"id":"W3212458632","doi":"10.5281/zenodo.3871016","title":"deferred correction software","year":2020,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; Software; Programming language","score_opus":0.05506760616288969,"score_gpt":0.24727118751585697,"score_spread":0.19220358135296728,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3212458632","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010518485,0.00025132496,0.8984742,0.00016215544,0.0003258913,0.00006394839,0.00072740007,0.094493404,0.004449903],"genre_scores_gemma":[0.08308714,0.00065615703,0.8043013,0.0007800801,0.00032547012,0.00041646432,0.003495601,0.067429624,0.03950806],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9976628,0.00043408087,0.00019147148,0.00035034763,0.0011820558,0.00017926219],"domain_scores_gemma":[0.9904974,0.003477377,0.00054279814,0.0024108838,0.0028262867,0.00024526735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023318012,0.0014478499,0.0010999978,0.0015424576,0.0007048989,0.0021516595,0.0039571268,0.0017428254,0.08575888],"category_scores_gemma":[0.020243013,0.0011565199,0.0012216427,0.0009122603,0.0008054306,0.0026175368,0.0024796452,0.0031032204,0.037426077],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007532789,0.0001456796,0.0014508882,0.0010967479,0.00018617288,0.0007132446,0.0005496135,0.034461014,0.02700348,0.08002672,0.24038215,0.61323094],"study_design_scores_gemma":[0.00033212808,0.0001573606,0.0011221281,0.00043093273,0.000107277665,0.0017028245,0.000105946165,0.3607128,0.08336241,0.099519394,0.4521211,0.00032572635],"about_ca_topic_score_codex":0.0020806023,"about_ca_topic_score_gemma":0.0021744135,"teacher_disagreement_score":0.08575888,"about_ca_system_score_codex":0.0006162643,"about_ca_system_score_gemma":0.0014543051,"threshold_uncertainty_score":0.28689206},"labels":[],"label_agreement":null},{"id":"W3215130818","doi":"10.1109/icsme52107.2021.00028","title":"Interactive Patch Filtering as Debugging Aid","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Debugging; Computer science; Correctness; Process (computing); Plug-in; Software engineering; Recall; Software bug; Precision and recall; Eclipse; Programming language; Human–computer interaction; Machine learning; Software","score_opus":0.016013490395001127,"score_gpt":0.27598801298719805,"score_spread":0.25997452259219694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3215130818","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08722401,0.0010435579,0.8490757,0.00055669446,0.00014040015,0.00032505335,0.00021665904,0.0571346,0.004283241],"genre_scores_gemma":[0.40881637,0.00037283858,0.5821902,0.00047744435,0.0000885066,0.00015467654,0.00046910945,0.0032630386,0.004167862],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9954052,0.0015893243,0.00032004598,0.00093488826,0.0014876504,0.0002629327],"domain_scores_gemma":[0.93522364,0.044937897,0.0031423608,0.010954118,0.0047912765,0.00095068885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005399574,0.0014367809,0.0011405483,0.0020297552,0.00052814453,0.0016313065,0.0025437835,0.0014483974,0.004696818],"category_scores_gemma":[0.033656843,0.0007319249,0.0007693992,0.00076086127,0.00073904207,0.0029030489,0.0020032637,0.0013907785,0.0013467015],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009831147,0.000702538,0.018420255,0.0010452438,0.00017664692,0.0008336745,0.002132653,0.011493769,0.084063634,0.0051570553,0.020736814,0.8542545],"study_design_scores_gemma":[0.0005438942,0.002510611,0.034556583,0.0007289855,0.0008564453,0.0063604508,0.0011118723,0.5708142,0.23480988,0.02252687,0.12476532,0.0004148001],"about_ca_topic_score_codex":0.0017357806,"about_ca_topic_score_gemma":0.0029825426,"teacher_disagreement_score":0.005399574,"about_ca_system_score_codex":0.00043779827,"about_ca_system_score_gemma":0.0008345975,"threshold_uncertainty_score":0.02855599},"labels":[],"label_agreement":null},{"id":"W3216180498","doi":"10.1109/icsme52107.2021.00026","title":"Mining Historical Test Failures to Dynamically Batch Tests to Save CI Resources","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Weighting; Metric (unit); Batch processing; Test case; Constant (computer programming); Test (biology); Machine learning; Programming language; Operations management; Engineering","score_opus":0.017222032460242877,"score_gpt":0.25771211069511063,"score_spread":0.24049007823486776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3216180498","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7131076,0.001352793,0.26270717,0.00056733505,0.00017206659,0.0004032992,0.00538903,0.013391597,0.0029091705],"genre_scores_gemma":[0.88681316,0.00024394674,0.10323661,0.00012488487,0.000049935206,0.00026052655,0.007694811,0.00051339704,0.001062669],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972446,0.0005398484,0.00024370391,0.0009021083,0.0008603809,0.00020945953],"domain_scores_gemma":[0.9818757,0.0077559054,0.0030275222,0.0034846002,0.0032240653,0.00063217484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037450432,0.0021264157,0.0011697755,0.0035695916,0.00043934298,0.0011902303,0.0028293368,0.00070903497,0.0010169036],"category_scores_gemma":[0.022434961,0.00077226694,0.0008414845,0.0024773735,0.00054218236,0.0022341285,0.0010063344,0.0012408722,0.00076971826],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076975365,0.0006467736,0.28524852,0.0006763149,0.00045144348,0.0005245188,0.00048204974,0.3478287,0.01880217,0.0022008377,0.016346969,0.3260219],"study_design_scores_gemma":[0.00006467517,0.00044130112,0.040036514,0.000054695665,0.00012144466,0.00025805508,0.00020555334,0.9394779,0.012098227,0.0033943567,0.0037926235,0.00005459675],"about_ca_topic_score_codex":0.010433215,"about_ca_topic_score_gemma":0.0152304815,"teacher_disagreement_score":0.010433215,"about_ca_system_score_codex":0.0009092557,"about_ca_system_score_gemma":0.0016255071,"threshold_uncertainty_score":0.02074498},"labels":[],"label_agreement":null},{"id":"W3216603549","doi":"10.1109/icsme52107.2021.00023","title":"Mutation Analysis for Assessing End-to-End Web Tests","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Test suite; Mutation testing; Source code; Web testing; Code coverage; Test case; Web application; Programmer; Web server; Suite; Static analysis; Web page; World Wide Web; Mutation; Software engineering; The Internet; Operating system; Web application security; Programming language; Web development; Software; Machine learning; Regression analysis","score_opus":0.0408933321761317,"score_gpt":0.3334186079602007,"score_spread":0.29252527578406895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3216603549","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2631586,0.0005746106,0.7257511,0.00014725084,0.00003582665,0.0005402597,0.00045344429,0.0071937353,0.0021451379],"genre_scores_gemma":[0.6900166,0.00012446419,0.3080483,0.0000631917,0.0000131385605,0.00033004695,0.0005756211,0.00026085842,0.0005677407],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99275804,0.0020606618,0.0005575033,0.0007789456,0.0034776106,0.0003671616],"domain_scores_gemma":[0.97957474,0.010726605,0.0033145582,0.0014816328,0.004422836,0.0004796891],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044088885,0.001449373,0.00083000393,0.0073365625,0.00061635737,0.0011454441,0.0016986212,0.0009906736,0.0010583492],"category_scores_gemma":[0.028117053,0.00030861923,0.0010472878,0.0016176745,0.0009798071,0.0015571247,0.0011533634,0.0009481957,0.00028803578],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007171187,0.0009042513,0.13434581,0.00065007224,0.00047601867,0.001340379,0.0009343746,0.2584907,0.12633012,0.0163811,0.0025351262,0.45689496],"study_design_scores_gemma":[0.00004859655,0.000671014,0.028248085,0.00011439947,0.00015828025,0.0007493494,0.00023468815,0.9107529,0.05016651,0.0063245543,0.0024237246,0.00010787005],"about_ca_topic_score_codex":0.004129169,"about_ca_topic_score_gemma":0.004003296,"teacher_disagreement_score":0.0073365625,"about_ca_system_score_codex":0.0012079346,"about_ca_system_score_gemma":0.0017147998,"threshold_uncertainty_score":0.023316681},"labels":[],"label_agreement":null},{"id":"W329547450","doi":"10.1007/978-3-319-00804-2_3","title":"Observations on Software Testing and Its Optimization","year":2013,"lang":"en","type":"book-chapter","venue":"Studies in computational intelligence","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Algoma University","funders":"","keywords":"Software testing; Software reliability testing; Computer science; Regression testing; System integration testing; Test strategy; Software; Coding (social sciences); Software construction; Non-regression testing; Software engineering; Software performance testing; Software development; Programming language; Mathematics","score_opus":0.24303127799320062,"score_gpt":0.35238518618001113,"score_spread":0.10935390818681051,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W329547450","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040969394,0.06349755,0.33963433,0.043254465,0.0015438237,0.000081255355,0.00079859135,0.0009152517,0.5093053],"genre_scores_gemma":[0.73626125,0.04556823,0.09744411,0.0052753673,0.004120491,0.00021202752,0.0006443964,0.0010407119,0.109433345],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984818,0.00048114467,0.0000997001,0.00024659687,0.000558632,0.00013212753],"domain_scores_gemma":[0.979392,0.017223101,0.0005482694,0.0016062085,0.0010425805,0.00018783307],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021070312,0.0011613186,0.0008550158,0.0015420846,0.00086099585,0.001709133,0.0014256611,0.0012029444,0.009551294],"category_scores_gemma":[0.016756736,0.0006042093,0.0006937535,0.002976562,0.006047407,0.0065977867,0.0012474153,0.006055768,0.0012972474],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006736681,0.000080990874,0.00041840156,0.00020100146,0.000012420472,0.00006994152,0.00022938257,0.010991505,0.0004152528,0.89663625,0.026974384,0.06390307],"study_design_scores_gemma":[0.000011805121,0.000019110705,0.0005784386,0.00012575736,0.000009016158,0.000077149714,0.00004855402,0.007995767,0.0007087929,0.9533673,0.037043005,0.000015361156],"about_ca_topic_score_codex":0.0051116967,"about_ca_topic_score_gemma":0.0032698384,"teacher_disagreement_score":0.009551294,"about_ca_system_score_codex":0.0019503188,"about_ca_system_score_gemma":0.00081813044,"threshold_uncertainty_score":0.031952262},"labels":[],"label_agreement":null},{"id":"W33337023","doi":"10.1007/978-0-387-35497-2_25","title":"Test Generation for CEFSM Combining Specification and Fault Coverage","year":2002,"lang":"en","type":"book-chapter","venue":"IFIP advances in information and communication technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"European Commission; Institut national de recherche en informatique et en automatique (INRIA)","keywords":"Nondeterministic algorithm; Computer science; Fault coverage; Reliability engineering; Code coverage; Cover (algebra); Fault (geology); Algorithm; Programming language; Engineering","score_opus":0.023605522039051273,"score_gpt":0.2605515167815273,"score_spread":0.23694599474247605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W33337023","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042625125,0.00015025242,0.9442186,0.00018578113,0.000049874252,0.00019642225,0.00043888294,0.0072510336,0.0048840176],"genre_scores_gemma":[0.4754149,0.000089381,0.5179433,0.00021483807,0.000037710397,0.00031114763,0.0016049432,0.0010191072,0.003364691],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99816316,0.0006246055,0.00011495733,0.00024071161,0.00067007064,0.00018650557],"domain_scores_gemma":[0.994822,0.0035668996,0.00021747689,0.0005651146,0.0007603571,0.00006798292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015685569,0.0011328247,0.00065003365,0.0024881943,0.00029019624,0.00074117206,0.0015924597,0.0014228582,0.005116406],"category_scores_gemma":[0.0069385134,0.00043355234,0.00086575036,0.00091053394,0.000681804,0.0010453794,0.000843438,0.0007615653,0.0009835922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010740084,0.00038344794,0.0045005498,0.0005576124,0.0001054137,0.00093163643,0.00023206358,0.21063219,0.07548265,0.028487295,0.010806104,0.666807],"study_design_scores_gemma":[0.00023956453,0.00033036896,0.0010136616,0.00006713517,0.00007137007,0.00053891685,0.0000388099,0.8690375,0.100531034,0.021984916,0.0061123194,0.00003442008],"about_ca_topic_score_codex":0.0019040549,"about_ca_topic_score_gemma":0.0020655084,"teacher_disagreement_score":0.005116406,"about_ca_system_score_codex":0.00064316986,"about_ca_system_score_gemma":0.0007533317,"threshold_uncertainty_score":0.01711607},"labels":[],"label_agreement":null},{"id":"W35884834","doi":"10.22215/etd/2009-10147","title":"An open framework for the specification and execution of a testable requirements model","year":2009,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science","score_opus":0.09014579382382956,"score_gpt":0.38785753413183016,"score_spread":0.2977117403080006,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W35884834","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00044653582,0.00006830655,0.9929178,0.00015131886,0.000040238803,0.00019412296,0.00019032054,0.0044401307,0.0015512911],"genre_scores_gemma":[0.024058366,0.00031875574,0.9681534,0.0001951308,0.000060632992,0.000781746,0.0014279464,0.0016626215,0.0033412678],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99232316,0.0023065438,0.0010651937,0.00095322664,0.002849171,0.0005026827],"domain_scores_gemma":[0.9896651,0.0052928557,0.0007019097,0.0026824067,0.0012368753,0.000420871],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013095406,0.0018531043,0.001433199,0.0027021386,0.0014434245,0.0080518015,0.0055845124,0.0036765782,0.012541808],"category_scores_gemma":[0.022760222,0.0024078977,0.004644109,0.0018471682,0.002897172,0.007268962,0.005875,0.00511338,0.004624948],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024277753,0.0002802951,0.00081807113,0.0009068639,0.0002005154,0.0011074342,0.0015464667,0.056251504,0.008089379,0.75117314,0.013877436,0.1655062],"study_design_scores_gemma":[0.000265834,0.00022625312,0.00032268526,0.0007856994,0.00021342892,0.000697754,0.00033935154,0.32369855,0.010365401,0.4021479,0.26077783,0.00015924682],"about_ca_topic_score_codex":0.006870197,"about_ca_topic_score_gemma":0.00838224,"teacher_disagreement_score":0.013095406,"about_ca_system_score_codex":0.00195638,"about_ca_system_score_gemma":0.0047857054,"threshold_uncertainty_score":0.06925595},"labels":[],"label_agreement":null},{"id":"W37109407","doi":"10.3390/life13040878","title":"An Ontology-based Software Test Generation Framework.","year":2010,"lang":"en","type":"article","venue":"Software Engineering and Knowledge Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"National Institute of Arthritis and Musculoskeletal and Skin Diseases; U.S. Department of Veterans Affairs","keywords":"Computer science; Test Management Approach; Software engineering; Test harness; Test case; Test (biology); Keyword-driven testing; Code coverage; Test suite; Manual testing; System under test; Reliability engineering; Software; Software development; Software construction; Programming language; Machine learning; Engineering","score_opus":0.010886739021814421,"score_gpt":0.24339811035775683,"score_spread":0.2325113713359424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W37109407","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018352978,0.00031609394,0.89210635,0.00040893408,0.00011710444,0.0008906325,0.005068341,0.096130095,0.0031271926],"genre_scores_gemma":[0.03949025,0.00079084845,0.92299,0.00044342843,0.000054522032,0.0017018291,0.024513675,0.006359399,0.003655962],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99567765,0.0011281228,0.0008204147,0.0007646295,0.0013259175,0.00028324962],"domain_scores_gemma":[0.99227285,0.0045464966,0.0005866551,0.0011895723,0.0011349041,0.00026956515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009799554,0.0030298824,0.0012402353,0.0050953203,0.0009087919,0.0038408665,0.005244115,0.0021169975,0.015389977],"category_scores_gemma":[0.015217743,0.0017623714,0.0065444745,0.0019005222,0.0012317725,0.003821974,0.0037417002,0.0028253589,0.0047134073],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007856904,0.0010118233,0.007921306,0.0053373906,0.0015118396,0.0022487708,0.0008727334,0.15930656,0.016969291,0.09469296,0.08455074,0.6247909],"study_design_scores_gemma":[0.000318157,0.00028808485,0.0013504061,0.0009159199,0.00042597257,0.00083636795,0.00024335085,0.7600714,0.013995797,0.10731092,0.114105955,0.00013767484],"about_ca_topic_score_codex":0.011032399,"about_ca_topic_score_gemma":0.014998268,"teacher_disagreement_score":0.015389977,"about_ca_system_score_codex":0.0019110325,"about_ca_system_score_gemma":0.0038316462,"threshold_uncertainty_score":0.051825643},"labels":[],"label_agreement":null},{"id":"W39437227","doi":"","title":"Model-Based Testing of Distributed Systems","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Correctness; Computer science; Reliability (semiconductor); Reliability engineering; Quality (philosophy); Engineering; Algorithm","score_opus":0.07862109517659838,"score_gpt":0.26056696555696235,"score_spread":0.18194587038036397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W39437227","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026142668,0.0022481503,0.95958287,0.0007275478,0.0000998582,0.000108340995,0.00011754227,0.0024144934,0.00855853],"genre_scores_gemma":[0.7645483,0.0017866977,0.22954252,0.00025264654,0.00012287506,0.0003021539,0.00046042816,0.00036097362,0.0026234568],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.9953414,0.002296562,0.00020451065,0.0003274597,0.0016287316,0.000201249],"domain_scores_gemma":[0.99430907,0.0040359073,0.00028170887,0.0008868418,0.00038352946,0.00010285931],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023884089,0.0008371654,0.00079165597,0.0009716469,0.00032099892,0.0019270363,0.0016175074,0.0010405133,0.0018460591],"category_scores_gemma":[0.011219544,0.00043922014,0.0008095974,0.0008159886,0.0013181393,0.0023859083,0.0012631039,0.0012525134,0.0003387065],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000238444,0.00029189762,0.0030573125,0.0006502355,0.0001662006,0.0005783041,0.00040601444,0.45702037,0.015460045,0.27898735,0.0053845146,0.23775935],"study_design_scores_gemma":[0.00006779245,0.00013025667,0.0005621432,0.00012872268,0.000032317836,0.00031070827,0.000045436453,0.8281325,0.008200595,0.15133406,0.01103216,0.000023279865],"about_ca_topic_score_codex":0.0018340853,"about_ca_topic_score_gemma":0.0011340178,"teacher_disagreement_score":0.0023884089,"about_ca_system_score_codex":0.0011334831,"about_ca_system_score_gemma":0.00084954314,"threshold_uncertainty_score":0.0126312375},"labels":[],"label_agreement":null},{"id":"W40826477","doi":"10.1007/978-0-387-35516-0_2","title":"Structural Coverage for Lotos","year":2000,"lang":"en","type":"book-chapter","venue":"IFIP advances in information and communication technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science","score_opus":0.010793855260251348,"score_gpt":0.2619111293880397,"score_spread":0.25111727412778834,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W40826477","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10599514,0.0019289149,0.7038756,0.0022771533,0.00024483583,0.00021786494,0.0016138067,0.0068780435,0.17696865],"genre_scores_gemma":[0.8916184,0.0011882144,0.071862526,0.00052748877,0.00028474507,0.00033780473,0.002256829,0.0013876116,0.030536396],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990195,0.00021358374,0.00006667074,0.00013241217,0.00041836099,0.00014937279],"domain_scores_gemma":[0.9967782,0.0019448141,0.00015601372,0.00062216277,0.00039943468,0.000099312674],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007076596,0.000610313,0.0006424222,0.0014243991,0.0009570514,0.0016618542,0.0008347231,0.00081381353,0.0151659],"category_scores_gemma":[0.004439288,0.0006272508,0.00067625486,0.0013507078,0.0011504282,0.0051260698,0.001696958,0.0014309591,0.002490816],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019266823,0.000060740917,0.0008680578,0.00031824416,0.000019758974,0.00021340261,0.00055867276,0.0091257645,0.0046904343,0.8352953,0.01487431,0.13378263],"study_design_scores_gemma":[0.000026426067,0.000056890258,0.00041858273,0.00008902746,0.000037846232,0.00032490533,0.000111901674,0.033109803,0.0046422207,0.9294458,0.03171452,0.000022120757],"about_ca_topic_score_codex":0.0006814669,"about_ca_topic_score_gemma":0.0008229845,"teacher_disagreement_score":0.0151659,"about_ca_system_score_codex":0.0007287904,"about_ca_system_score_gemma":0.00053768087,"threshold_uncertainty_score":0.050734997},"labels":[],"label_agreement":null},{"id":"W4104734","doi":"10.1007/978-3-642-39742-4_26","title":"A Multi-objective Genetic Algorithm for Generating Test Suites from Extended Finite State Machines","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Consortium de Recherche et d’innovation en Aérospatiale au Québec","keywords":"Computer science; Test suite; Finite-state machine; Genetic algorithm; Similarity (geometry); Algorithm; Set (abstract data type); State (computer science); Test (biology); Suite; Test case; Artificial intelligence; Machine learning; Programming language","score_opus":0.020592682319030896,"score_gpt":0.2616508539926293,"score_spread":0.2410581716735984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4104734","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024845554,0.00019951105,0.9708773,0.0000958964,0.00003787613,0.00015566459,0.00008584318,0.0012294727,0.0024729439],"genre_scores_gemma":[0.19801143,0.000117985524,0.79888844,0.000097479875,0.000022847053,0.00045596834,0.00030891597,0.0002121968,0.001884829],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994373,0.00018384565,0.000026366726,0.00010724819,0.00018526636,0.00005998015],"domain_scores_gemma":[0.9984226,0.0012173987,0.00008248671,0.000061677136,0.00017594076,0.000039930208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010840441,0.001228133,0.001196775,0.0014896394,0.0004903472,0.00069712405,0.0016517713,0.0017432004,0.0028566753],"category_scores_gemma":[0.0034327542,0.00068362907,0.0012848385,0.0011197436,0.0007360108,0.00063830503,0.0009337185,0.0013145831,0.00040387042],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005917088,0.00009593178,0.00037382316,0.000066050634,0.00005773331,0.00008548695,0.000046531404,0.87448996,0.0023028282,0.0042066397,0.0008358107,0.117379986],"study_design_scores_gemma":[0.000016581096,0.000025043204,0.00006959585,0.000006523242,0.000009777391,0.000012495472,0.0000036661127,0.9980258,0.0003628763,0.0013247811,0.00013918911,0.000003676273],"about_ca_topic_score_codex":0.0070237876,"about_ca_topic_score_gemma":0.006551438,"teacher_disagreement_score":0.0070237876,"about_ca_system_score_codex":0.0010590077,"about_ca_system_score_gemma":0.0014208028,"threshold_uncertainty_score":0.0139657855},"labels":[],"label_agreement":null},{"id":"W413907436","doi":"10.1016/j.scico.2014.09.005","title":"Crawl-based analysis of web applications: Prospects and challenges","year":2014,"lang":"en","type":"article","venue":"Science of Computer Programming","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Sketch; Web crawler; Computer science; Crawling; World Wide Web; Web application; Web testing; Field (mathematics); Data science; Web service; Web application security; Web development","score_opus":0.022321936001959024,"score_gpt":0.26792613935501,"score_spread":0.24560420335305094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W413907436","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3041019,0.046298184,0.56912196,0.032298144,0.0008537605,0.00036095583,0.0026267872,0.03068502,0.013653279],"genre_scores_gemma":[0.71563137,0.011034013,0.26163656,0.0015044896,0.000811073,0.0001892239,0.0027769625,0.0016618585,0.0047544874],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98949933,0.0039434643,0.0005099199,0.0011667636,0.004406086,0.00047449055],"domain_scores_gemma":[0.94139427,0.028639428,0.0037782365,0.0106459595,0.0136716645,0.0018704259],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077660056,0.0012253993,0.0020235567,0.009487138,0.0010542184,0.0059190462,0.0042114467,0.0024877554,0.0012200848],"category_scores_gemma":[0.029748522,0.0007809541,0.00087986776,0.006902207,0.0017004019,0.009859185,0.0017525285,0.0020526063,0.0013659779],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003018892,0.0010652948,0.08686421,0.00089482183,0.00037014423,0.00035692728,0.0007733665,0.014620423,0.016436901,0.015140559,0.030116014,0.83305943],"study_design_scores_gemma":[0.000086786145,0.00045526627,0.04655755,0.0006161086,0.0002795422,0.0016271471,0.0021940675,0.7542414,0.0396543,0.10985837,0.044201143,0.00022823404],"about_ca_topic_score_codex":0.0047878637,"about_ca_topic_score_gemma":0.0068897814,"teacher_disagreement_score":0.009487138,"about_ca_system_score_codex":0.0012545908,"about_ca_system_score_gemma":0.0030442118,"threshold_uncertainty_score":0.041071057},"labels":[],"label_agreement":null},{"id":"W41990731","doi":"","title":"Knowledge-based Software Test Generation.","year":2009,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Test Management Approach; Computer science; Test harness; Keyword-driven testing; Test (biology); Test script; Software engineering; Manual testing; Test case; Unit testing; System under test; Ontology; White-box testing; Reliability engineering; Software; Software construction; Software development; Programming language; Machine learning; Engineering","score_opus":0.03646622812087495,"score_gpt":0.2826508291638791,"score_spread":0.24618460104300416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W41990731","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010382661,0.0003197002,0.97332406,0.0004274018,0.00005645547,0.0005419257,0.00025746564,0.005275177,0.009415102],"genre_scores_gemma":[0.30263802,0.00033391683,0.69010967,0.00042313742,0.000037361675,0.0005601593,0.0013258269,0.000492128,0.0040797973],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9957461,0.0020446915,0.00019551745,0.00047819305,0.0013538444,0.00018178769],"domain_scores_gemma":[0.98358744,0.0117116645,0.00068851194,0.0025110242,0.001321554,0.00017975952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049254447,0.0006651116,0.00045183796,0.0015213875,0.00029820143,0.0013632677,0.0017218393,0.0014099926,0.006656504],"category_scores_gemma":[0.0309686,0.00036118232,0.0006934161,0.0009457591,0.0007812246,0.0016562839,0.001755311,0.000938028,0.0019415092],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021076249,0.00046391616,0.0029421058,0.00043236214,0.00009090424,0.00045058198,0.0003524737,0.051439382,0.010003316,0.03843351,0.009660203,0.8855204],"study_design_scores_gemma":[0.00018351687,0.00028305355,0.002755951,0.0003117994,0.00009199285,0.0010620047,0.00016973764,0.8516181,0.027144445,0.08820006,0.028116196,0.00006314038],"about_ca_topic_score_codex":0.0015207466,"about_ca_topic_score_gemma":0.0020855265,"teacher_disagreement_score":0.006656504,"about_ca_system_score_codex":0.00078275485,"about_ca_system_score_gemma":0.0009682199,"threshold_uncertainty_score":0.026048541},"labels":[],"label_agreement":null},{"id":"W4200053401","doi":"10.1109/models-c53483.2021.00095","title":"MRegTest: A Replay-Based Regression Testing Tool for Distributed UML-RT Models","year":2021,"lang":"en","type":"article","venue":"2021 ACM/IEEE International Conference on Model Driven Engineering Languages and Systems Companion (MODELS-C)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Regression testing; Semantics (computer science); Model-based testing; Process (computing); Set (abstract data type); Unified Modeling Language; Timestamp; Regression analysis; Data mining; Test case; Programming language; Machine learning; Real-time computing; Software; Software system","score_opus":0.11174798614387775,"score_gpt":0.32266939813125173,"score_spread":0.21092141198737396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200053401","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023810951,0.00022325332,0.77892506,0.0001909494,0.000061492756,0.00016207257,0.0012639335,0.19322294,0.0021393457],"genre_scores_gemma":[0.35433394,0.0003865359,0.611622,0.0002725789,0.000040865027,0.00066600094,0.0048409007,0.022852272,0.0049849814],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99803406,0.0005879599,0.00017394121,0.00036501756,0.0007251702,0.00011390732],"domain_scores_gemma":[0.99387866,0.0039350796,0.00066570845,0.00089909614,0.00052553916,0.00009591081],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002288193,0.0017540257,0.0007135029,0.0017600567,0.00032348398,0.0008919278,0.0024181688,0.0012059192,0.007049445],"category_scores_gemma":[0.01050503,0.0008040687,0.0013181331,0.00050638826,0.00066224724,0.002012942,0.0014162922,0.0014350954,0.0017795094],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011817107,0.00084532227,0.018606387,0.0020721797,0.00048923894,0.0031395836,0.0019225333,0.25117442,0.08310943,0.024925703,0.058710232,0.55382323],"study_design_scores_gemma":[0.00020922913,0.0003652443,0.0031122188,0.0002583741,0.00011874756,0.0011803451,0.00017503252,0.87471175,0.06701407,0.010705571,0.04201209,0.00013731413],"about_ca_topic_score_codex":0.0029605422,"about_ca_topic_score_gemma":0.0030411498,"teacher_disagreement_score":0.007049445,"about_ca_system_score_codex":0.0005269716,"about_ca_system_score_gemma":0.0009356377,"threshold_uncertainty_score":0.023582757},"labels":[],"label_agreement":null},{"id":"W4200439617","doi":"10.18280/ijsse.110606","title":"Qualitative Analysis of State/Event Fault Trees Based on Interface Automata","year":2021,"lang":"en","type":"article","venue":"International Journal of Safety and Security Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Fundamental Research Funds for the Central Universities; Nanjing University of Aeronautics and Astronautics; National Natural Science Foundation of China","keywords":"Automaton; Computer science; Interface (matter); Fault tree analysis; Semantics (computer science); Event (particle physics); Theoretical computer science; Tree (set theory); Set (abstract data type); Process (computing); State (computer science); Finite-state machine; Algorithm; Programming language; Mathematics; Reliability engineering; Engineering","score_opus":0.013979296515769313,"score_gpt":0.3259175296679685,"score_spread":0.3119382331521992,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200439617","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024158372,0.000033662327,0.97428054,0.00002627459,0.000005942202,0.000049066668,0.00007073116,0.00036234083,0.0010131667],"genre_scores_gemma":[0.7314005,0.0001147043,0.26691115,0.00003768051,0.000009581376,0.00020812004,0.00023860055,0.0000901657,0.0009894634],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991522,0.00018416341,0.000056742636,0.00013471884,0.0003589684,0.000113278045],"domain_scores_gemma":[0.997792,0.0012219361,0.0002768318,0.00017531074,0.00046718153,0.00006669458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008924336,0.0005312853,0.0003188885,0.0018105126,0.00037689306,0.00090581493,0.0006781081,0.00045141907,0.0018732919],"category_scores_gemma":[0.00309568,0.00020774775,0.0010153261,0.0006382825,0.0014787074,0.0015823856,0.0005888546,0.0007127744,0.00015301866],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020758189,0.00011997573,0.0068857516,0.00028330775,0.00007463948,0.00070595474,0.00081946515,0.5337715,0.049640764,0.34482256,0.00061589153,0.062052585],"study_design_scores_gemma":[0.000012249796,0.000057175494,0.0007915847,0.00002675609,0.000029131648,0.000106182175,0.0000862187,0.90043586,0.009685943,0.0875912,0.0011581859,0.000019531426],"about_ca_topic_score_codex":0.0033581357,"about_ca_topic_score_gemma":0.0016661232,"teacher_disagreement_score":0.0033581357,"about_ca_system_score_codex":0.0010009148,"about_ca_system_score_gemma":0.0009801904,"threshold_uncertainty_score":0.0072621703},"labels":[],"label_agreement":null},{"id":"W4205529760","doi":"10.1145/3502297","title":"Automated, Cost-effective, and Update-driven App Testing","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Centre National de la Recherche Scientifique; European Commission","keywords":"Computer science; Dependability; Oracle; Code coverage; Random testing; Android (operating system); Model-based testing; Test case; Cyclomatic complexity; Automation; Source code; Code (set theory); Test suite; Random oracle; Set (abstract data type); Software engineering; Distributed computing; Programming language; Machine learning; Software; Operating system; Encryption","score_opus":0.07249387697246641,"score_gpt":0.3145700562029212,"score_spread":0.24207617923045482,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4205529760","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15705416,0.0014665993,0.8192918,0.0007029131,0.000061269944,0.0006426671,0.00042658296,0.015524265,0.0048298026],"genre_scores_gemma":[0.76754314,0.00033325196,0.22929019,0.00020029784,0.00003506151,0.0003196809,0.0006312526,0.0005701125,0.0010769832],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99109155,0.0033184458,0.00043463722,0.0008976819,0.0038285137,0.00042908857],"domain_scores_gemma":[0.97000265,0.01731408,0.0023362862,0.0074348277,0.0024849675,0.00042728084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002919322,0.0017710633,0.0009345435,0.0029032836,0.00053294405,0.0016592053,0.0032552117,0.0015475117,0.0018451982],"category_scores_gemma":[0.025245955,0.0007453322,0.0012695156,0.0011310703,0.0012171279,0.0034729457,0.002508893,0.0013167345,0.0007290658],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006953966,0.0009925532,0.0249818,0.0008876346,0.00028825575,0.0007789396,0.000841223,0.21976617,0.06923053,0.011543869,0.0048825303,0.6651111],"study_design_scores_gemma":[0.00012228572,0.00084165623,0.006297393,0.00013791726,0.0002141048,0.00096718495,0.00025169842,0.91972655,0.048994407,0.01738507,0.004977495,0.00008426043],"about_ca_topic_score_codex":0.0033490767,"about_ca_topic_score_gemma":0.005227069,"teacher_disagreement_score":0.0033490767,"about_ca_system_score_codex":0.0010048145,"about_ca_system_score_gemma":0.0026046678,"threshold_uncertainty_score":0.0154390335},"labels":[],"label_agreement":null},{"id":"W4213240734","doi":"10.2197/ipsjjip.30.155","title":"Automatic Optimize-time Validation for Binary Optimizers","year":2022,"lang":"en","type":"article","venue":"Journal of Information Processing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada)","funders":"","keywords":"Computer science; Binary number; Compatibility (geochemistry); Code (set theory); Binary code; Code coverage; Program optimization; Source code; Dead code; Redundant code; Programming language; Code generation; Set (abstract data type); Operating system; Software; Arithmetic","score_opus":0.01353373715607724,"score_gpt":0.2568205463823833,"score_spread":0.24328680922630602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4213240734","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.082964346,0.0005748293,0.8142666,0.0003408225,0.0002129387,0.00040916272,0.00044988442,0.096644446,0.0041369926],"genre_scores_gemma":[0.51448673,0.00018798368,0.46566287,0.0005835468,0.00009211758,0.00050381647,0.0017283541,0.012791767,0.0039628125],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98821414,0.0035207437,0.0010656219,0.0015936086,0.0045856074,0.0010202345],"domain_scores_gemma":[0.9696978,0.010447012,0.0041865613,0.010059334,0.0052082036,0.00040114459],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045612743,0.002026228,0.0011810737,0.002149395,0.0007700238,0.0018951218,0.0028999285,0.0013002985,0.004225542],"category_scores_gemma":[0.024375677,0.0014348474,0.0015602598,0.000952537,0.0013437797,0.0035461765,0.0022833496,0.002111157,0.0016967576],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035383203,0.00094165164,0.034376487,0.0014066285,0.0004174531,0.00084034976,0.0010275071,0.06402336,0.22383447,0.020567141,0.024369316,0.62465733],"study_design_scores_gemma":[0.00030099545,0.0008796665,0.0090104565,0.00023656967,0.00023180402,0.00078477245,0.0001226845,0.5781445,0.36983955,0.012085964,0.028073017,0.00028996816],"about_ca_topic_score_codex":0.0021765365,"about_ca_topic_score_gemma":0.0023432204,"teacher_disagreement_score":0.0045612743,"about_ca_system_score_codex":0.0013480566,"about_ca_system_score_gemma":0.0024730335,"threshold_uncertainty_score":0.024122596},"labels":[],"label_agreement":null},{"id":"W4213241976","doi":"10.1109/ms.2021.3133805","title":"AI-Driven Development Is Here: Should You Worry?","year":2022,"lang":"en","type":"article","venue":"IEEE Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Worry; Computer science; Software development; Software engineering; Software; Psychology; Operating system","score_opus":0.04386800979763472,"score_gpt":0.2825543218158692,"score_spread":0.23868631201823448,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4213241976","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0003470702,0.013333796,0.0018725132,0.9639807,0.016856275,0.000015714966,0.000041181676,0.00023473265,0.0033179964],"genre_scores_gemma":[0.022142984,0.046283677,0.010170537,0.8555783,0.04176672,0.000111938134,0.00019971884,0.0005952787,0.023150917],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9888487,0.004027194,0.00075827586,0.0011325981,0.004185563,0.0010475939],"domain_scores_gemma":[0.91590756,0.03558434,0.0040251357,0.005171298,0.02906677,0.010244889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021497782,0.0011433853,0.0015499096,0.0017622766,0.0054010414,0.012976945,0.0023799774,0.012733931,0.018441055],"category_scores_gemma":[0.08190932,0.0005814435,0.0009057631,0.001660133,0.011319607,0.032499243,0.0046813893,0.034871,0.013624879],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009025289,0.00011035997,0.0018905337,0.0006328692,0.00006058709,0.00026293923,0.0011484835,0.00012966346,0.00030313054,0.030755075,0.85205454,0.112561636],"study_design_scores_gemma":[0.00006152246,0.00012080601,0.0013926394,0.0026598093,0.00005990812,0.0009602432,0.0059678038,0.00037787997,0.00034528837,0.09531817,0.8926252,0.00011062012],"about_ca_topic_score_codex":0.005925567,"about_ca_topic_score_gemma":0.00872375,"teacher_disagreement_score":0.021497782,"about_ca_system_score_codex":0.002740672,"about_ca_system_score_gemma":0.008804598,"threshold_uncertainty_score":0.11369252},"labels":[],"label_agreement":null},{"id":"W4214654833","doi":"10.26226/morressier.5ada8a0ed462b8029238e544","title":"The Value Of A Specimen Tracking Tool And Beyond","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Value (mathematics); Tracking (education); Computer science; Psychology; Machine learning","score_opus":0.02511246415375651,"score_gpt":0.2778520837146284,"score_spread":0.2527396195608719,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214654833","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06496919,0.074206546,0.547845,0.17877859,0.027819168,0.0016261797,0.0052085286,0.01573661,0.08381018],"genre_scores_gemma":[0.21671788,0.022159481,0.6783098,0.040889185,0.008014187,0.0018840093,0.0034610226,0.0033428303,0.025221618],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98347396,0.006905498,0.0012271383,0.002349631,0.0056606373,0.00038310006],"domain_scores_gemma":[0.8742694,0.059150834,0.007964832,0.02763514,0.025776759,0.005203087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.045091078,0.001201069,0.0010799062,0.0035506226,0.0014955122,0.0072052637,0.0036443712,0.0037355195,0.013681664],"category_scores_gemma":[0.07647299,0.0007148015,0.001007226,0.001681269,0.004217794,0.013913882,0.0058414172,0.0041796435,0.008252882],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011503197,0.0001940759,0.022260012,0.0014162558,0.00010514346,0.00078846695,0.0019614063,0.0010127872,0.010210481,0.025268689,0.08106966,0.8545627],"study_design_scores_gemma":[0.0001175352,0.0013111444,0.018985724,0.006489153,0.00025154962,0.005520129,0.0020906897,0.0061834306,0.01881855,0.062817395,0.87702596,0.00038874144],"about_ca_topic_score_codex":0.0016304277,"about_ca_topic_score_gemma":0.0014000383,"teacher_disagreement_score":0.045091078,"about_ca_system_score_codex":0.0021878027,"about_ca_system_score_gemma":0.004954089,"threshold_uncertainty_score":0.23846728},"labels":[],"label_agreement":null},{"id":"W4220798833","doi":"10.18280/isi.270106","title":"Automatic Generation and Optimization of Combinatorial Test Cases from UML Activity Diagram Using Particle Swarm Optimization","year":2022,"lang":"en","type":"article","venue":"Ingénierie des systèmes d information","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Activity diagram; Test case; Particle swarm optimization; Computer science; Model-based testing; Unified Modeling Language; Process (computing); Automation; Test Management Approach; Combinatorial explosion; Test (biology); Software; Algorithm; Data mining; Software system; Programming language; Machine learning; Mathematics; Engineering; Regression analysis","score_opus":0.02835137652987759,"score_gpt":0.24920017531039684,"score_spread":0.22084879878051925,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220798833","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05090328,0.00010050713,0.9440414,0.000094484545,0.00002553489,0.00023210005,0.00009720669,0.0023616555,0.0021439076],"genre_scores_gemma":[0.42628345,0.00013130804,0.5713875,0.000044327728,0.000007550961,0.0004591021,0.00045927355,0.00017441454,0.0010530612],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990891,0.00030392958,0.000075169475,0.0001348417,0.00031304907,0.000083965555],"domain_scores_gemma":[0.9975672,0.0017236659,0.00023200875,0.00014021619,0.00028203984,0.000054864275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010305719,0.0012451564,0.0008081075,0.0017057849,0.00030995172,0.0008008704,0.0007306036,0.00073811476,0.0017440493],"category_scores_gemma":[0.0036136105,0.00062463275,0.0009701401,0.00069874927,0.0004429171,0.00047018105,0.0005138098,0.0005958033,0.00023793908],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000108017884,0.00013913996,0.0019101034,0.00016361938,0.000055437118,0.00022181764,0.00009275881,0.8712224,0.016030451,0.002933548,0.0008170389,0.10630574],"study_design_scores_gemma":[0.000020495234,0.000046003148,0.0002738923,0.000006033977,0.000009702911,0.000026770444,0.000010070858,0.99546367,0.0032298004,0.0005277301,0.00038068526,0.0000052453006],"about_ca_topic_score_codex":0.0043409537,"about_ca_topic_score_gemma":0.0033504711,"teacher_disagreement_score":0.0043409537,"about_ca_system_score_codex":0.00065312115,"about_ca_system_score_gemma":0.0008480763,"threshold_uncertainty_score":0.008631349},"labels":[],"label_agreement":null},{"id":"W4220988444","doi":"10.1109/tse.2022.3162236","title":"Selecting Context-Sensitivity Modularly for Accelerating Object-Sensitive Pointer Analysis","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Notation; Pointer (user interface); Pointer analysis; Computer science; Programming language; Context (archaeology); Theoretical computer science; Algorithm; Mathematics; Static analysis; Artificial intelligence; Arithmetic","score_opus":0.019674976284096056,"score_gpt":0.23646358962900285,"score_spread":0.2167886133449068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220988444","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03563252,0.00060597627,0.93242145,0.0003419876,0.00014041971,0.00019838323,0.00011650988,0.023167111,0.0073756864],"genre_scores_gemma":[0.28831822,0.00058103196,0.6974962,0.0007766374,0.00013113495,0.00031186204,0.00042807017,0.005351834,0.006605095],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99883705,0.00019824877,0.00009048816,0.0002804849,0.00040153248,0.00019226791],"domain_scores_gemma":[0.99726105,0.0010589613,0.00022663883,0.0009534893,0.00038943544,0.00011038555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011559201,0.0014759016,0.0007449268,0.0011011709,0.0006196414,0.0020635563,0.0022517487,0.00076743675,0.006261737],"category_scores_gemma":[0.0059754737,0.00072352146,0.0010134258,0.0009194915,0.001339947,0.00336849,0.003178358,0.0021012563,0.0029480893],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083806604,0.00032949945,0.007828146,0.00083960884,0.00016276202,0.0008303908,0.0016421538,0.030381147,0.23681204,0.11563222,0.018272497,0.58643144],"study_design_scores_gemma":[0.00016929732,0.00033799213,0.0027561777,0.00034161966,0.00025958833,0.0007360967,0.0003533719,0.41738036,0.398335,0.089403756,0.08969143,0.0002353336],"about_ca_topic_score_codex":0.0013644047,"about_ca_topic_score_gemma":0.0025797444,"teacher_disagreement_score":0.006261737,"about_ca_system_score_codex":0.0008265781,"about_ca_system_score_gemma":0.0021797162,"threshold_uncertainty_score":0.020947576},"labels":[],"label_agreement":null},{"id":"W4225425474","doi":"10.3389/feduc.2022.853578","title":"Using Content Coding and Automatic Item Generation to Improve Test Security","year":2022,"lang":"en","type":"article","venue":"Frontiers in Education","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Coding (social sciences); Scalability; Item bank; Process (computing); Data mining; Item response theory; Database","score_opus":0.04183309971773125,"score_gpt":0.2869789934573029,"score_spread":0.24514589373957163,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225425474","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023594134,0.00007308664,0.96783656,0.00029245336,0.000057324803,0.0011251435,0.00014651137,0.0036268644,0.0032478524],"genre_scores_gemma":[0.093668275,0.00007633987,0.9029536,0.00010486727,0.000022306565,0.00092058664,0.00040814275,0.00051370094,0.0013321143],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.96478903,0.024023581,0.0018256212,0.0018900498,0.00679339,0.00067836326],"domain_scores_gemma":[0.8414865,0.09835453,0.005919343,0.024141459,0.029297646,0.0008005316],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02166215,0.0015941656,0.0008297944,0.0087439995,0.0010981368,0.0027808812,0.002410192,0.001293442,0.0050143413],"category_scores_gemma":[0.14212456,0.00087117637,0.0008200251,0.006215573,0.0022395838,0.00538716,0.0038721303,0.001748793,0.0021227591],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040732077,0.00053733686,0.00932591,0.00037251983,0.00007171881,0.00027356838,0.0054264707,0.013490193,0.01270761,0.030928519,0.0074484716,0.91901046],"study_design_scores_gemma":[0.00057356857,0.0013229565,0.017249715,0.0013452312,0.00021111622,0.0017392391,0.0046556303,0.61261755,0.11994844,0.19571789,0.044128727,0.00049001613],"about_ca_topic_score_codex":0.0024974025,"about_ca_topic_score_gemma":0.0023357337,"teacher_disagreement_score":0.02166215,"about_ca_system_score_codex":0.0024140934,"about_ca_system_score_gemma":0.0034727529,"threshold_uncertainty_score":0.11456174},"labels":[],"label_agreement":null},{"id":"W4225845951","doi":"10.1109/qrs-c55045.2021.00080","title":"Boosting Grey-box Fuzzing for Connected Autonomous Vehicle Systems","year":2021,"lang":"en","type":"article","venue":"2021 IEEE 21st International Conference on Software Quality, Reliability and Security Companion (QRS-C)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ross Video (Canada); Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Fuzz testing; Computer science; Software; Automotive industry; Symbolic execution; Concolic testing; Code coverage; Vulnerability (computing); Process (computing); Computer security; Embedded system; Artificial intelligence; Programming language; Engineering","score_opus":0.07344966310209487,"score_gpt":0.33783700142671286,"score_spread":0.26438733832461797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225845951","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22312069,0.0003436038,0.7688028,0.00030444804,0.00004420356,0.00012933131,0.00004987598,0.002200725,0.005004288],"genre_scores_gemma":[0.92819,0.00009175379,0.07030126,0.0000774087,0.000011704311,0.00003328008,0.00005118594,0.000072571944,0.0011708766],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99884784,0.00023891879,0.000049597624,0.00021572907,0.00050674897,0.00014107264],"domain_scores_gemma":[0.99668485,0.0022606065,0.00022235763,0.00029947812,0.00044008554,0.00009265878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016698772,0.0007158342,0.00060265674,0.0010969192,0.0005314603,0.0010065857,0.0009845017,0.00081427896,0.001894447],"category_scores_gemma":[0.007741437,0.0003932744,0.00086016924,0.000348289,0.001611784,0.0015588157,0.0013694696,0.0010299855,0.00019678139],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034655083,0.000102307946,0.0036156916,0.00012280606,0.000071312745,0.00027644885,0.00032844488,0.838894,0.028662074,0.044085372,0.0005839366,0.08291095],"study_design_scores_gemma":[0.000011845711,0.000058015932,0.00019554383,0.0000110287565,0.000011450879,0.000025890591,0.000010241368,0.9802612,0.006533852,0.012469528,0.00040374903,0.00000770084],"about_ca_topic_score_codex":0.0066753435,"about_ca_topic_score_gemma":0.005453675,"teacher_disagreement_score":0.0066753435,"about_ca_system_score_codex":0.0013533747,"about_ca_system_score_gemma":0.00090189226,"threshold_uncertainty_score":0.013272941},"labels":[],"label_agreement":null},{"id":"W4226139682","doi":"10.5281/zenodo.6415365","title":"Scalable and Accurate Test Case Prioritization in Continuous Integration Contexts","year":2022,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of Ottawa","funders":"","keywords":"Prioritization; Scalability; Test (biology); Computer science; Biology; Engineering; Database; Ecology; Process management","score_opus":0.028062328126160885,"score_gpt":0.2553470633048817,"score_spread":0.22728473517872083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226139682","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15674967,0.004665816,0.039875105,0.0012561831,0.00064268603,0.00089693343,0.71680236,0.06568692,0.013424363],"genre_scores_gemma":[0.09421998,0.00053097977,0.03447165,0.0002777603,0.00012656707,0.0006381633,0.8661042,0.0018033693,0.0018273704],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9928215,0.0017258046,0.00078622,0.0022087486,0.0019312903,0.0005264245],"domain_scores_gemma":[0.97905934,0.008321185,0.0013959204,0.00638214,0.0038004483,0.0010410001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004614701,0.002968885,0.0014283124,0.006381895,0.00084923673,0.0032931406,0.0036113004,0.0017685673,0.0066172164],"category_scores_gemma":[0.02685704,0.0007874847,0.001587844,0.006146225,0.00065360527,0.0033383952,0.0030055572,0.0020488454,0.006567898],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015387558,0.0014270013,0.04583643,0.0036632787,0.00047423964,0.0006386566,0.0005128741,0.028479345,0.006240874,0.004352451,0.7445155,0.16232067],"study_design_scores_gemma":[0.0015506359,0.0013024575,0.11386604,0.0015444764,0.0005138241,0.0018494873,0.0015456863,0.27396232,0.023869887,0.027846018,0.55184704,0.00030210512],"about_ca_topic_score_codex":0.0071628396,"about_ca_topic_score_gemma":0.013040538,"teacher_disagreement_score":0.0071628396,"about_ca_system_score_codex":0.0015033723,"about_ca_system_score_gemma":0.0025781742,"threshold_uncertainty_score":0.024405122},"labels":[],"label_agreement":null},{"id":"W4226176814","doi":"10.1007/978-3-030-99527-0_19","title":"Automatic Repair for Network Programs","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Debugging; Modular design; Reuse; Abstraction; Set (abstract data type); Domain (mathematical analysis); Symbolic execution; Task (project management); Programming language; Software; Software engineering; Distributed computing; Embedded system; Systems engineering","score_opus":0.02667669564335165,"score_gpt":0.2650263496974982,"score_spread":0.23834965405414657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226176814","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.068858944,0.00034193986,0.9109875,0.000155324,0.000032976597,0.0001041193,0.00020443823,0.015468067,0.0038465979],"genre_scores_gemma":[0.556252,0.00028109047,0.43536648,0.00010042955,0.00002452893,0.000114079325,0.0008165234,0.0016124005,0.0054324646],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99901617,0.0002593881,0.00006245306,0.00020802305,0.00035783218,0.00009610577],"domain_scores_gemma":[0.9966947,0.0015630018,0.00044951346,0.0008725314,0.0003832707,0.000036985755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008247037,0.000734909,0.00048608487,0.0010550382,0.0003748439,0.00066545274,0.0015524094,0.00057849573,0.0040432126],"category_scores_gemma":[0.0045149378,0.00028416552,0.00049272995,0.0006342823,0.0007996636,0.0012358198,0.0009353533,0.00070188736,0.0005208582],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003927082,0.00014973903,0.0031206822,0.0008897787,0.00006236298,0.00073734927,0.00077258545,0.14596048,0.092500634,0.03891587,0.008727099,0.7077707],"study_design_scores_gemma":[0.000059952312,0.00041216277,0.002026469,0.00016575576,0.00008509703,0.0010976725,0.00021717815,0.73520726,0.18771136,0.044646632,0.028319845,0.00005065303],"about_ca_topic_score_codex":0.0017794584,"about_ca_topic_score_gemma":0.0017295,"teacher_disagreement_score":0.0040432126,"about_ca_system_score_codex":0.0007040365,"about_ca_system_score_gemma":0.00071096775,"threshold_uncertainty_score":0.013525903},"labels":[],"label_agreement":null},{"id":"W4226466002","doi":"10.1109/qrs54544.2021.00093","title":"MINTS: Unsupervised Temporal Specifications Miner","year":2021,"lang":"en","type":"article","venue":"2021 IEEE 21st International Conference on Software Quality, Reliability and Security (QRS)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Correctness; Computer science; Debugging; Task (project management); Finite-state machine; Scalability; Frame (networking); Software; State (computer science); Trie; Software bug; Software system; Artificial intelligence; Programming language; Data structure; Engineering; Database; Systems engineering","score_opus":0.10583280632417456,"score_gpt":0.34160529800087,"score_spread":0.2357724916766954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226466002","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015949756,0.00045549485,0.9462593,0.0003698213,0.00007815745,0.00025313886,0.00740912,0.027345328,0.001879897],"genre_scores_gemma":[0.17979737,0.00033031646,0.7839303,0.00038183975,0.00005944866,0.0006453912,0.02884185,0.0020130074,0.0040004845],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99885786,0.00024216599,0.00012434658,0.00038594048,0.00031332002,0.0000764204],"domain_scores_gemma":[0.99676263,0.0019364712,0.00022627943,0.0005612992,0.00043284774,0.00008043925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012350332,0.0010578255,0.0006226752,0.0016372615,0.00052356825,0.0011055293,0.0023333502,0.0011231262,0.004875531],"category_scores_gemma":[0.008825713,0.0007936501,0.0017598062,0.0012187585,0.0006062924,0.002005195,0.001717199,0.0015164193,0.0022906784],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007922871,0.0003329186,0.012530496,0.0013483153,0.00048196394,0.000893466,0.0005652016,0.24362981,0.018460877,0.039686743,0.06480631,0.61647165],"study_design_scores_gemma":[0.00005965423,0.000070949456,0.0005817048,0.000051993247,0.000035798254,0.00031685916,0.0000845832,0.93472177,0.0090290215,0.039113827,0.015911816,0.000022078948],"about_ca_topic_score_codex":0.0034104509,"about_ca_topic_score_gemma":0.013291674,"teacher_disagreement_score":0.004875531,"about_ca_system_score_codex":0.00075681176,"about_ca_system_score_gemma":0.0028263074,"threshold_uncertainty_score":0.016310334},"labels":[],"label_agreement":null},{"id":"W4229062148","doi":"10.1016/j.infsof.2022.106936","title":"A search-based framework for automatic generation of testing environments for cyber–physical systems","year":2022,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Cyber-physical system; Drone; Computer science; Robot; Software; Virtual machine; Physical system; Distributed computing; Human–computer interaction; Systems engineering; Embedded system; Real-time computing; Artificial intelligence; Engineering; Operating system","score_opus":0.03980151792648982,"score_gpt":0.2772495253795939,"score_spread":0.2374480074531041,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4229062148","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003489503,0.00010728178,0.98822355,0.00008059984,0.00001476444,0.00010678601,0.00016946097,0.0070982883,0.00070972287],"genre_scores_gemma":[0.13233358,0.00008767239,0.8646081,0.000117989366,0.000023386428,0.00023858796,0.0007963296,0.0008344627,0.00095994154],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978555,0.0005776909,0.0001655968,0.00033281866,0.00087206846,0.00019630272],"domain_scores_gemma":[0.9952236,0.0028630388,0.00029046534,0.00064478203,0.0008323642,0.0001457824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002056244,0.0012880714,0.0013199471,0.0022975556,0.00086522877,0.0018255666,0.0028396025,0.0017631416,0.0055633816],"category_scores_gemma":[0.00939452,0.0007692882,0.0017447826,0.0013551571,0.0012466183,0.0023427233,0.0022818248,0.0015748589,0.0015333921],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085735193,0.00046507627,0.002935529,0.0009501558,0.0002161259,0.0006869653,0.00046263705,0.3804464,0.030982666,0.056172695,0.0147600025,0.5110644],"study_design_scores_gemma":[0.00005697441,0.000065111075,0.00019428514,0.00003347737,0.000031551637,0.00009423928,0.000028126913,0.96956694,0.005890082,0.021500869,0.0025202981,0.000018081368],"about_ca_topic_score_codex":0.0058018197,"about_ca_topic_score_gemma":0.009559941,"teacher_disagreement_score":0.0058018197,"about_ca_system_score_codex":0.0009942519,"about_ca_system_score_gemma":0.0022390394,"threshold_uncertainty_score":0.018611372},"labels":[],"label_agreement":null},{"id":"W4230318508","doi":"10.22215/etd/2010-08842","title":"Method and tool support for refinement of test suites","year":2010,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Library and Archives Canada","funders":"","keywords":"Test (biology); Computer science; Geology","score_opus":0.017882349998801794,"score_gpt":0.33241269764019243,"score_spread":0.31453034764139065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4230318508","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026136078,0.00016423083,0.96939194,0.000123308,0.00006455829,0.00017050502,0.00041491975,0.02493356,0.002123361],"genre_scores_gemma":[0.07368805,0.00030994188,0.90878725,0.00016567558,0.00009332032,0.000623325,0.002302492,0.007127249,0.006902777],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98974353,0.0031518855,0.0014570209,0.0011451266,0.0039651473,0.00053728005],"domain_scores_gemma":[0.96719956,0.017553683,0.0013703408,0.008534321,0.004798432,0.0005437273],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010660666,0.001628358,0.0012917897,0.0046287775,0.0008233089,0.0035045026,0.0038959957,0.0016300397,0.01958106],"category_scores_gemma":[0.043509163,0.001337394,0.0026513175,0.0022054561,0.00084231526,0.0033716543,0.0027106986,0.0020659203,0.006290358],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078766106,0.00035053727,0.0051562074,0.0011250501,0.00033189848,0.0011193694,0.0009866282,0.022328854,0.025889168,0.05954453,0.034552682,0.8478273],"study_design_scores_gemma":[0.00088396174,0.00045084325,0.0031850587,0.0009561265,0.0003627137,0.0028905678,0.00022650017,0.5283583,0.0925974,0.08726761,0.28248692,0.00033397452],"about_ca_topic_score_codex":0.0021166268,"about_ca_topic_score_gemma":0.0027592883,"teacher_disagreement_score":0.01958106,"about_ca_system_score_codex":0.0007583097,"about_ca_system_score_gemma":0.0023106511,"threshold_uncertainty_score":0.06550515},"labels":[],"label_agreement":null},{"id":"W4231515836","doi":"10.1109/jcdl.2017.7991565","title":"A Text Extraction Software Benchmark Based on a Synthesized Dataset","year":2017,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Vienna Science and Technology Fund","keywords":"Computer science; Ground truth; Correctness; Workflow; Information retrieval; Data mining; Benchmark (surveying); Scalability; Quality (philosophy); Process (computing); Snippet; Artificial intelligence; Database; Algorithm; Programming language","score_opus":0.03252267041055522,"score_gpt":0.31590586622439903,"score_spread":0.2833831958138438,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231515836","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4540379,0.002490315,0.31902766,0.0019366535,0.00078092696,0.0054639215,0.16354793,0.037781358,0.014933355],"genre_scores_gemma":[0.27907833,0.00079691096,0.35487503,0.00053819484,0.00009465767,0.0043020677,0.35542434,0.0012821537,0.0036082815],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99707055,0.0008796686,0.00043534004,0.0007270491,0.00074544235,0.00014184476],"domain_scores_gemma":[0.991405,0.0035915105,0.0005254441,0.0017119397,0.002526224,0.00023982607],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002790733,0.0012670567,0.0005177315,0.0028921706,0.00078553084,0.0012521341,0.002121509,0.0016218169,0.0018056621],"category_scores_gemma":[0.010904677,0.00032517876,0.0012286725,0.0027795844,0.00073115947,0.00142694,0.0012913629,0.0011328188,0.0012112303],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020529958,0.0047215684,0.032192487,0.0068659796,0.0006903075,0.0027977754,0.0014118507,0.23087782,0.08005082,0.018616552,0.21097653,0.40874538],"study_design_scores_gemma":[0.0012091454,0.002642853,0.040905457,0.0007086412,0.0003403932,0.0024732612,0.0014248955,0.59318244,0.14807299,0.016809855,0.19192918,0.00030099673],"about_ca_topic_score_codex":0.0044921488,"about_ca_topic_score_gemma":0.0063961735,"teacher_disagreement_score":0.0044921488,"about_ca_system_score_codex":0.0015174254,"about_ca_system_score_gemma":0.0013541972,"threshold_uncertainty_score":0.014759004},"labels":[],"label_agreement":null},{"id":"W4231608201","doi":"10.22215/etd/2009-09375","title":"A genetic algorithm to devise test case sequences based on a state machine diagram and data flow information","year":2009,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage; Library and Archives Canada","funders":"","keywords":"Diagram; Computer science; Algorithm; Data flow diagram; State (computer science); Information flow; State diagram; Combinatorics; Mathematics; Philosophy; Database","score_opus":0.01779673111461176,"score_gpt":0.2920255916742575,"score_spread":0.27422886055964574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231608201","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015898352,0.00008303767,0.9793687,0.00009746864,0.000031213604,0.0002604824,0.00007055917,0.0018169606,0.002373268],"genre_scores_gemma":[0.1289701,0.00008176745,0.8684707,0.00009116843,0.000012951118,0.00038970137,0.0002619824,0.00015688386,0.0015646736],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999355,0.00018454279,0.000040668547,0.000153407,0.00021127671,0.000055140106],"domain_scores_gemma":[0.9975473,0.0017068608,0.00016471352,0.000118816235,0.0003913929,0.00007090364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011428028,0.0010502903,0.0008508158,0.0027482782,0.00053432706,0.00086546666,0.0014765377,0.0013108834,0.0032930407],"category_scores_gemma":[0.005784087,0.0006221537,0.00077360385,0.0012355326,0.00076158816,0.0006901665,0.0006212285,0.0010454884,0.0006979736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015142988,0.00034521666,0.002984645,0.00016851757,0.00016584927,0.00021333144,0.00022246527,0.5559146,0.010179566,0.011718749,0.0029000985,0.41503558],"study_design_scores_gemma":[0.000049616054,0.000080012855,0.00033811486,0.000028206472,0.000043786433,0.00007420247,0.000022418932,0.99293715,0.0021566297,0.0031668646,0.0010920016,0.000011038557],"about_ca_topic_score_codex":0.012061254,"about_ca_topic_score_gemma":0.008550906,"teacher_disagreement_score":0.012061254,"about_ca_system_score_codex":0.0010502327,"about_ca_system_score_gemma":0.002235045,"threshold_uncertainty_score":0.023982108},"labels":[],"label_agreement":null},{"id":"W4232005383","doi":"10.1109/icse.2015.52","title":"Detecting Inconsistencies in JavaScript MVC Applications","year":2015,"lang":"en","type":"article","venue":"2015 IEEE/ACM 37th IEEE International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Intel Corporation","keywords":"Computer science; JavaScript; Web application; Consistency (knowledge bases); Model–view–controller; Identifier; Data mining; Software engineering; Programming language; Distributed computing; Operating system; Artificial intelligence; User interface","score_opus":0.10755743350707492,"score_gpt":0.32242141870734803,"score_spread":0.2148639852002731,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4232005383","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5369575,0.0013134648,0.39193416,0.00038780057,0.00015940779,0.0003601259,0.0012524087,0.06502583,0.0026093281],"genre_scores_gemma":[0.7535995,0.00021816083,0.2416386,0.0002070197,0.00004009782,0.00009455229,0.0020973973,0.0012071626,0.0008975227],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9883024,0.0018644445,0.0011416527,0.0020135012,0.0062592356,0.00041875395],"domain_scores_gemma":[0.9568609,0.01964762,0.007662263,0.005726584,0.009511099,0.00059147493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004850116,0.0011161871,0.0011795801,0.0053419913,0.0006421351,0.0017726,0.0021693087,0.0015819932,0.0005985884],"category_scores_gemma":[0.03490975,0.0007353713,0.0005299516,0.0024218746,0.000601038,0.001767687,0.0019107688,0.0010686427,0.00045782322],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011474037,0.00085376116,0.22286929,0.0012503477,0.00036889763,0.0025865787,0.0037235036,0.03418497,0.09951077,0.00494579,0.013671123,0.61488754],"study_design_scores_gemma":[0.0001181492,0.000565104,0.073474705,0.00028228655,0.00022541097,0.0029786914,0.00083433604,0.7541371,0.14496797,0.0056445743,0.016562961,0.00020874897],"about_ca_topic_score_codex":0.0045905854,"about_ca_topic_score_gemma":0.0048195133,"teacher_disagreement_score":0.0053419913,"about_ca_system_score_codex":0.0008586183,"about_ca_system_score_gemma":0.0012088467,"threshold_uncertainty_score":0.025650203},"labels":[],"label_agreement":null},{"id":"W4232175613","doi":"10.1007/978-3-642-35182-2_17","title":"Concurrent Test Generation Using Concolic Multi-trace Analysis","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Concolic testing; Computer science; Thread (computing); Concurrency; Code coverage; Programming language; Symbolic execution; TRACE (psycholinguistics); Test case; Parallel computing; Software; Machine learning","score_opus":0.07016234621544544,"score_gpt":0.3164600363995561,"score_spread":0.24629769018411068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4232175613","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025477802,0.000054293032,0.96384,0.00007429405,0.000049266444,0.00020095237,0.00007621838,0.007195163,0.003031998],"genre_scores_gemma":[0.50693387,0.00004452113,0.4870943,0.00011575457,0.000043731274,0.00023721822,0.0003463791,0.0015561432,0.00362807],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969416,0.0006812559,0.00016866834,0.00058667193,0.0012564976,0.00036529222],"domain_scores_gemma":[0.9895281,0.005257574,0.0005819218,0.0022323679,0.0021343962,0.00026569577],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018502815,0.0013539514,0.0007285605,0.0022956838,0.0007097986,0.0011018625,0.0026274244,0.0011609282,0.0055686417],"category_scores_gemma":[0.008754922,0.0006796411,0.001104293,0.0011726643,0.0010943745,0.001709692,0.002018967,0.0012763573,0.0009701337],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013366902,0.0007621462,0.0064704106,0.00051502703,0.00018884422,0.0014781383,0.0005678408,0.122002475,0.08079152,0.055860594,0.0067281695,0.72329813],"study_design_scores_gemma":[0.00018390779,0.00033701753,0.0007598785,0.000056084322,0.00011883767,0.00060057297,0.000045793113,0.87252176,0.08115163,0.039738044,0.004437274,0.00004932197],"about_ca_topic_score_codex":0.002236713,"about_ca_topic_score_gemma":0.0028974076,"teacher_disagreement_score":0.0055686417,"about_ca_system_score_codex":0.000742315,"about_ca_system_score_gemma":0.0016697063,"threshold_uncertainty_score":0.018629014},"labels":[],"label_agreement":null},{"id":"W4232229135","doi":"10.1109/ibf50092.2020.9034868","title":"Message from the Chairs","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; World Wide Web","score_opus":0.03640098502946325,"score_gpt":0.23401821829736166,"score_spread":0.19761723326789843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4232229135","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00051057525,0.0023027414,0.0008344468,0.46176404,0.50754476,0.00019293613,0.00089942047,0.00062259723,0.02532855],"genre_scores_gemma":[0.0057224925,0.0022633113,0.0011530548,0.42955244,0.07952108,0.0005196969,0.0009259843,0.00059145724,0.47975048],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99726415,0.00047129882,0.00015542326,0.00048447528,0.0011029587,0.0005217093],"domain_scores_gemma":[0.98621225,0.001104187,0.00035162212,0.00042718585,0.006922913,0.004981767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041322527,0.001349307,0.001295008,0.00090860046,0.0040511773,0.0065071913,0.0017151273,0.011762829,0.18464193],"category_scores_gemma":[0.019453837,0.0007408748,0.0013623657,0.0006430403,0.00090285175,0.004202657,0.0036686324,0.019170344,0.15041472],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029360726,0.00001073057,0.0000505614,0.000020798141,0.0000017936361,0.00002667284,0.000017178516,0.000009424441,0.00005700069,0.00036973736,0.99613035,0.0032763632],"study_design_scores_gemma":[0.000018978953,0.000028463137,0.00031375617,0.000058520876,0.0000057128045,0.000053101296,0.00017845248,0.000044559416,0.0001299772,0.0004478711,0.9987012,0.000019443207],"about_ca_topic_score_codex":0.005090517,"about_ca_topic_score_gemma":0.0076144543,"teacher_disagreement_score":0.18464193,"about_ca_system_score_codex":0.0024309556,"about_ca_system_score_gemma":0.006135335,"threshold_uncertainty_score":0.6176888},"labels":[],"label_agreement":null},{"id":"W4233095787","doi":"10.22215/etd/2012-07238","title":"Comparison of coverage criteria for the category partition method using automatically generated test suites with melba","year":2012,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Library and Archives Canada","funders":"","keywords":"Partition (number theory); Computer science; Mathematics; Information retrieval; Combinatorics","score_opus":0.083910690938459,"score_gpt":0.4105903065037051,"score_spread":0.3266796155652461,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4233095787","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7730884,0.0030548824,0.20264947,0.00034284111,0.00012278039,0.00035062377,0.001668333,0.009057257,0.009665355],"genre_scores_gemma":[0.8830297,0.00025831672,0.11188873,0.00006647587,0.000019073064,0.00025187337,0.0027446635,0.00063669746,0.0011044424],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9923901,0.0028920213,0.0006729325,0.00074048835,0.0028258252,0.00047865685],"domain_scores_gemma":[0.9454238,0.04476602,0.0016315845,0.0025873794,0.005077456,0.0005136947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054275426,0.0011478623,0.0011594126,0.007078463,0.0004591634,0.0016025909,0.0018954581,0.0015816061,0.002480825],"category_scores_gemma":[0.032339267,0.00039421677,0.0008977109,0.0025539633,0.00058341277,0.0014068366,0.0010032046,0.00058693904,0.00045980397],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007021465,0.0011417364,0.046137497,0.0014953667,0.00081807456,0.0004929452,0.0008926135,0.2869949,0.05627217,0.0080732135,0.0078118495,0.5828482],"study_design_scores_gemma":[0.00032963546,0.0012543718,0.023438444,0.00013113617,0.00020226413,0.0004467689,0.0003211732,0.93807876,0.029810354,0.0023889032,0.0035176657,0.000080450045],"about_ca_topic_score_codex":0.008250662,"about_ca_topic_score_gemma":0.010711879,"teacher_disagreement_score":0.008250662,"about_ca_system_score_codex":0.0014884591,"about_ca_system_score_gemma":0.0013211218,"threshold_uncertainty_score":0.028703928},"labels":[],"label_agreement":null},{"id":"W4233113820","doi":"10.22215/etd/2009-10188","title":"Using machine learning to refine black-box test specifications and test suites","year":2009,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage; Library and Archives Canada","funders":"","keywords":"Test (biology); Computer science; Artificial intelligence","score_opus":0.056099875847924155,"score_gpt":0.3105859834633714,"score_spread":0.25448610761544727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4233113820","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039621156,0.00026631032,0.950203,0.00017950252,0.000039553383,0.00017315002,0.00025269866,0.0078414995,0.0014230409],"genre_scores_gemma":[0.34734327,0.00019796271,0.64713275,0.00014766834,0.000033660865,0.00022617023,0.0022019385,0.0009364515,0.0017800663],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9937418,0.0024152058,0.00048667094,0.0009897064,0.001929339,0.0004372997],"domain_scores_gemma":[0.95827085,0.027552187,0.0030684217,0.004640864,0.006103979,0.00036368315],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004783707,0.0017171508,0.0014163138,0.0044860803,0.0005696236,0.001988023,0.0019387032,0.0011006115,0.0034546922],"category_scores_gemma":[0.038658895,0.0010124168,0.0019819527,0.0013939092,0.0013622326,0.0027287328,0.001388212,0.0019476912,0.0009069247],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030851038,0.0003458555,0.0061436323,0.0003445863,0.00018315786,0.000324811,0.00027433463,0.55116993,0.010384123,0.010868971,0.0024711178,0.417181],"study_design_scores_gemma":[0.000024996127,0.000054199743,0.00044485534,0.000048961378,0.000025389103,0.000043532033,0.000023980221,0.9829464,0.0051004603,0.010426602,0.0008490145,0.000011591832],"about_ca_topic_score_codex":0.008622234,"about_ca_topic_score_gemma":0.014244357,"teacher_disagreement_score":0.008622234,"about_ca_system_score_codex":0.0018369814,"about_ca_system_score_gemma":0.0027000823,"threshold_uncertainty_score":0.025298953},"labels":[],"label_agreement":null},{"id":"W4233672025","doi":"10.22215/etd/2010-08757","title":"Hybrid strength two covering array constructions: using cover starters to create covering arrays","year":2010,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Heritage; Library and Archives Canada","funders":"","keywords":"Cover (algebra); Computer science; Humanities; Art; Engineering; Mechanical engineering","score_opus":0.016809667172424964,"score_gpt":0.2838758492318568,"score_spread":0.2670661820594319,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4233672025","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07216162,0.0001452905,0.9051163,0.00013881498,0.00013306769,0.00013001333,0.0001250054,0.003142793,0.018907089],"genre_scores_gemma":[0.40657064,0.00022180422,0.5730637,0.00015894238,0.000058199916,0.000244324,0.00046084955,0.0017382016,0.017483294],"study_design_codex":"bench_or_experimental","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989818,0.000206505,0.000052631032,0.00018168888,0.00042198072,0.0001554054],"domain_scores_gemma":[0.99732614,0.0010176887,0.00020367268,0.0008617019,0.00043213507,0.00015868052],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063503266,0.0007116178,0.00045029173,0.0008517364,0.0006060708,0.0011885806,0.0009240281,0.00075457065,0.008578288],"category_scores_gemma":[0.002797473,0.00058932253,0.00063426874,0.0008621151,0.00096597214,0.0022660692,0.0019586587,0.0009317194,0.002295649],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007075337,0.00026419584,0.0024327715,0.00041803188,0.000079690624,0.001290716,0.0015267832,0.045128983,0.38063788,0.21364029,0.009194984,0.34467813],"study_design_scores_gemma":[0.00014073167,0.0010189791,0.0014258673,0.00010943378,0.000107629436,0.0015893143,0.00045915326,0.18417475,0.6153826,0.092458546,0.10300781,0.00012517293],"about_ca_topic_score_codex":0.000295406,"about_ca_topic_score_gemma":0.00050345913,"teacher_disagreement_score":0.008578288,"about_ca_system_score_codex":0.00041664953,"about_ca_system_score_gemma":0.0004053067,"threshold_uncertainty_score":0.028697252},"labels":[],"label_agreement":null},{"id":"W4234480656","doi":"10.1145/1108792.1108795","title":"An empirical framework for comparing effectiveness of testing and property-based formal analysis","year":2005,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Debugging; Computer science; Software engineering; Software bug; Formal methods; Random testing; Formal specification; Empirical research; Formal verification; Measure (data warehouse); Software testing; Software; Programming language; Data mining; Test case; Machine learning; Mathematics","score_opus":0.06517471039457418,"score_gpt":0.3506701034153422,"score_spread":0.28549539302076804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4234480656","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35656852,0.0055076014,0.58812976,0.0027132044,0.00039304202,0.0061804685,0.0025702857,0.00052790967,0.037409175],"genre_scores_gemma":[0.8713796,0.0006809516,0.11817151,0.0005576054,0.00022565336,0.0071614315,0.0010246744,0.00016249296,0.0006361884],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.7122272,0.22735792,0.013474599,0.0062917513,0.03882209,0.0018263763],"domain_scores_gemma":[0.16266471,0.7729352,0.020989612,0.024957076,0.017056806,0.0013965473],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.16449411,0.0013838521,0.0014274794,0.009356646,0.0010555172,0.003996932,0.0020725715,0.0026485275,0.0025146191],"category_scores_gemma":[0.5871231,0.0004318826,0.001746516,0.0069868304,0.0072557987,0.007953887,0.0034275986,0.0029865366,0.0005548862],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006897229,0.0057490077,0.33759263,0.004243935,0.004459649,0.0004281285,0.0069778073,0.052395497,0.0068265055,0.27912837,0.0061576967,0.28914356],"study_design_scores_gemma":[0.0028462573,0.03516693,0.3187849,0.002949649,0.0021248797,0.002025178,0.0068847844,0.29610234,0.013213503,0.2852987,0.03396941,0.0006334405],"about_ca_topic_score_codex":0.0009305927,"about_ca_topic_score_gemma":0.0004897852,"teacher_disagreement_score":0.16449411,"about_ca_system_score_codex":0.0020824396,"about_ca_system_score_gemma":0.0018208545,"threshold_uncertainty_score":0.86993843},"labels":[],"label_agreement":null},{"id":"W4234836192","doi":"10.1145/2345141.1967693","title":"Software debugging and testing using the abstract diagnosis theory","year":2012,"lang":"en","type":"article","venue":"ACM SIGPLAN Notices","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Observability; Debugging; Dependency (UML); Program slicing; Controllability; Variable (mathematics); Set (abstract data type); Source code; A priori and a posteriori; Algorithm; Programming language; Theoretical computer science; Mathematics","score_opus":0.08703010010577043,"score_gpt":0.3037358530034371,"score_spread":0.21670575289766664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4234836192","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032180117,0.00032038076,0.99427265,0.00031834806,0.00003237777,0.000054739223,0.000047581216,0.00022018795,0.0015158359],"genre_scores_gemma":[0.22798774,0.0015238263,0.7665526,0.00042592484,0.00030171472,0.00037642152,0.0004350843,0.000117897565,0.0022787508],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9952833,0.0015658046,0.0003394174,0.0008685291,0.0016292036,0.0003137039],"domain_scores_gemma":[0.9912322,0.0067778802,0.0005679194,0.0007765099,0.00049659394,0.00014883504],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027099808,0.0017384181,0.0015086232,0.0029496846,0.00089147285,0.0026994017,0.001789856,0.001760645,0.0033796658],"category_scores_gemma":[0.010917738,0.0007918869,0.0027772645,0.0018122697,0.004341347,0.005986995,0.0028380598,0.003802406,0.00056694215],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015555181,0.00012159946,0.0010447638,0.00057459,0.000108314336,0.00033935727,0.0003023182,0.25892746,0.0048778895,0.60330135,0.0020527388,0.12819415],"study_design_scores_gemma":[0.00005711504,0.000085015825,0.00019976862,0.00008624649,0.00004323296,0.00013368785,0.000050725295,0.4927818,0.0020542834,0.500015,0.0044668945,0.000026236725],"about_ca_topic_score_codex":0.0031459793,"about_ca_topic_score_gemma":0.0018393575,"teacher_disagreement_score":0.0033796658,"about_ca_system_score_codex":0.002216937,"about_ca_system_score_gemma":0.0022741498,"threshold_uncertainty_score":0.016085088},"labels":[],"label_agreement":null},{"id":"W4236669998","doi":"10.1002/stvr.401","title":"Modelling methods for web application verification and testing: state of the art","year":2008,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Web testing; Software engineering; Software; Web application; Verification and validation; Web modeling; Software testing; Data mining; World Wide Web; Web service; Programming language; Engineering","score_opus":0.06046944475577618,"score_gpt":0.3077181642846764,"score_spread":0.24724871952890023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4236669998","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022149896,0.029323732,0.9630346,0.00042306102,0.00008295322,0.00008165259,0.000058156747,0.0008743589,0.0039064987],"genre_scores_gemma":[0.13574481,0.05162431,0.80689394,0.00032033774,0.0004370692,0.00046643146,0.00055336254,0.00059134274,0.0033684538],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98904306,0.004453625,0.000701218,0.0008807558,0.00470952,0.00021190345],"domain_scores_gemma":[0.9779936,0.016656462,0.0011332243,0.0023761117,0.0016784834,0.00016210652],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0070502628,0.0018626117,0.0021252148,0.0053510773,0.0005704704,0.0042960313,0.003863483,0.0026575564,0.0043265405],"category_scores_gemma":[0.017430214,0.0013564231,0.0025010079,0.005152005,0.0032176848,0.005006043,0.002005405,0.0025396757,0.002087678],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000102984966,0.0001964529,0.0016082665,0.0035997499,0.00031214353,0.00021212717,0.00042255115,0.07489446,0.0041866307,0.22736327,0.0044045327,0.68269676],"study_design_scores_gemma":[0.000074735144,0.000092941205,0.000831822,0.001986859,0.00013723249,0.0004824691,0.0001533318,0.5906611,0.005176758,0.31984955,0.080430984,0.00012218965],"about_ca_topic_score_codex":0.0028793486,"about_ca_topic_score_gemma":0.0010959003,"teacher_disagreement_score":0.9929497,"about_ca_system_score_codex":0.0013370902,"about_ca_system_score_gemma":0.0012345415,"threshold_uncertainty_score":0.037285805},"labels":[],"label_agreement":null},{"id":"W4237956829","doi":"10.1145/2016603.1967693","title":"Software debugging and testing using the abstract diagnosis theory","year":2011,"lang":"en","type":"article","venue":"ACM SIGPLAN Notices","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Observability; Debugging; Dependency (UML); Program slicing; Controllability; Variable (mathematics); Source code; Set (abstract data type); A priori and a posteriori; Context (archaeology); Algorithm; Programming language; Theoretical computer science; Mathematics","score_opus":0.13478969035681715,"score_gpt":0.28714426921478553,"score_spread":0.15235457885796838,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4237956829","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032180117,0.00032038076,0.99427265,0.00031834806,0.00003237777,0.000054739223,0.000047581216,0.00022018795,0.0015158359],"genre_scores_gemma":[0.22798774,0.0015238263,0.7665526,0.00042592484,0.00030171472,0.00037642152,0.0004350843,0.000117897565,0.0022787508],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9952833,0.0015658046,0.0003394174,0.0008685291,0.0016292036,0.0003137039],"domain_scores_gemma":[0.9912322,0.0067778802,0.0005679194,0.0007765099,0.00049659394,0.00014883504],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027099808,0.0017384181,0.0015086232,0.0029496846,0.00089147285,0.0026994017,0.001789856,0.001760645,0.0033796658],"category_scores_gemma":[0.010917738,0.0007918869,0.0027772645,0.0018122697,0.004341347,0.005986995,0.0028380598,0.003802406,0.00056694215],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015555181,0.00012159946,0.0010447638,0.00057459,0.000108314336,0.00033935727,0.0003023182,0.25892746,0.0048778895,0.60330135,0.0020527388,0.12819415],"study_design_scores_gemma":[0.00005711504,0.000085015825,0.00019976862,0.00008624649,0.00004323296,0.00013368785,0.000050725295,0.4927818,0.0020542834,0.500015,0.0044668945,0.000026236725],"about_ca_topic_score_codex":0.0031459793,"about_ca_topic_score_gemma":0.0018393575,"teacher_disagreement_score":0.0033796658,"about_ca_system_score_codex":0.002216937,"about_ca_system_score_gemma":0.0022741498,"threshold_uncertainty_score":0.016085088},"labels":[],"label_agreement":null},{"id":"W4238745643","doi":"10.22215/etd/2006-06634","title":"Traffic-aware stress testing of distributed real-time systems based on UML models using genetic algorithms","year":2006,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage","funders":"","keywords":"Computer science; Unified Modeling Language; Algorithm; Programming language; Software","score_opus":0.03484226928523558,"score_gpt":0.2733626140505468,"score_spread":0.2385203447653112,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4238745643","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40530503,0.00024678835,0.587654,0.00030363488,0.000033799235,0.00008467633,0.00006370934,0.002785868,0.0035225507],"genre_scores_gemma":[0.9355019,0.000057736906,0.06366364,0.000020941936,0.0000050062977,0.000044775254,0.00006841261,0.00007012613,0.0005673915],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99925953,0.00035072968,0.000031432235,0.00007625191,0.00021434689,0.000067736146],"domain_scores_gemma":[0.99572206,0.0030438344,0.0004283586,0.00028438718,0.00042100827,0.00010027316],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082922523,0.0007679639,0.00043382004,0.0009802871,0.00022121993,0.00096161815,0.0009655695,0.00070862053,0.0008817649],"category_scores_gemma":[0.00516753,0.00033100523,0.00066649046,0.00030656176,0.0004935424,0.00096979627,0.00043599264,0.0004318881,0.00010322833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015263996,0.00014671305,0.0032649308,0.000052568805,0.000050750565,0.00010102094,0.00014029797,0.9403975,0.0105554955,0.004393673,0.00023565187,0.040508766],"study_design_scores_gemma":[0.000009148145,0.000021195843,0.00018527222,0.0000054626853,0.000009261305,0.0000103975135,0.000010764963,0.99623716,0.002297141,0.0011246973,0.00008687552,0.0000025892914],"about_ca_topic_score_codex":0.00517378,"about_ca_topic_score_gemma":0.0051104673,"teacher_disagreement_score":0.00517378,"about_ca_system_score_codex":0.0010337672,"about_ca_system_score_gemma":0.0009045312,"threshold_uncertainty_score":0.010287344},"labels":[],"label_agreement":null},{"id":"W4239722293","doi":"10.1002/9780470382844.gloss","title":"Glossary","year":2008,"lang":"en","type":"other","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Glossary; Library science; Engineering; Citation; Sociology; Computer science; Philosophy","score_opus":0.018116398211921975,"score_gpt":0.24258748086514478,"score_spread":0.2244710826532228,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4239722293","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001178038,0.023472246,0.0074017085,0.008671629,0.009737532,0.0010145651,0.18501529,0.0022089477,0.7613],"genre_scores_gemma":[0.016872738,0.042870235,0.016676227,0.013888707,0.004737433,0.0020676951,0.3107526,0.0033789042,0.58875555],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99845386,0.00031729462,0.00030077514,0.00032537305,0.00046143512,0.00014126035],"domain_scores_gemma":[0.99628806,0.0011810209,0.0002633017,0.0005438253,0.0015129327,0.00021082626],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0017455252,0.0015750043,0.0014391916,0.0056387153,0.002750027,0.0048662326,0.0028631128,0.0020040628,0.5374914],"category_scores_gemma":[0.013619728,0.00061326346,0.0009527394,0.0077225817,0.0010365542,0.004982714,0.0022846796,0.0021083571,0.3931765],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003822531,0.0000147643195,0.00023487676,0.00071553525,0.000007619658,0.00012300402,0.00017065127,0.00007976919,0.00010563045,0.008280869,0.95048445,0.039744593],"study_design_scores_gemma":[0.0000041035432,0.0000050455915,0.00020605106,0.00034766924,0.000004721213,0.000100780846,0.00010498336,0.000019933887,0.000029762989,0.001534838,0.99763596,0.000006179047],"about_ca_topic_score_codex":0.018929612,"about_ca_topic_score_gemma":0.013615897,"teacher_disagreement_score":0.46250862,"about_ca_system_score_codex":0.0025172087,"about_ca_system_score_gemma":0.0032555566,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4240999823","doi":"10.1109/cmsbse.2013.6605713","title":"Effectively using search-based software engineering techniques within model checking and its applications","year":2013,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Search-based software engineering; Assertion; Software engineering; Software; Process (computing); Model checking; Heuristic; Software development; Software construction; Programming language; Artificial intelligence","score_opus":0.030477762737352792,"score_gpt":0.2658187234315718,"score_spread":0.23534096069421903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4240999823","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007808572,0.0004914295,0.98869544,0.0005890303,0.000020093115,0.00005779628,0.000015636746,0.0005066459,0.0018153648],"genre_scores_gemma":[0.2321167,0.00095512817,0.76536983,0.00025661293,0.000034855868,0.00013597621,0.00006637773,0.00020123713,0.0008632057],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99129933,0.005593338,0.00042605575,0.00042913717,0.002055949,0.00019622353],"domain_scores_gemma":[0.9716635,0.023287462,0.0008342225,0.0029494606,0.0011296808,0.00013568734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073750634,0.0011255372,0.0011651985,0.0025653546,0.0007385788,0.002308714,0.0018685013,0.001538325,0.0020647654],"category_scores_gemma":[0.026224563,0.0008581042,0.0017136094,0.0023815178,0.002182406,0.0048022014,0.0025938067,0.0021767602,0.00059034006],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024019933,0.00034469264,0.0041744085,0.0010772005,0.0004245853,0.000307265,0.0009032322,0.45410421,0.016604658,0.20140582,0.0014045974,0.31900904],"study_design_scores_gemma":[0.000051280396,0.0001621635,0.00028922808,0.00021015799,0.00012024771,0.00021610451,0.00013894595,0.85630065,0.008369511,0.1289736,0.0051331795,0.000034948516],"about_ca_topic_score_codex":0.0033465454,"about_ca_topic_score_gemma":0.0052772295,"teacher_disagreement_score":0.0073750634,"about_ca_system_score_codex":0.00096952013,"about_ca_system_score_gemma":0.00231974,"threshold_uncertainty_score":0.03900349},"labels":[],"label_agreement":null},{"id":"W4242156770","doi":"10.1109/icse.2015.78","title":"DASE: Document-Assisted Symbolic Execution for Improving Automated Software Testing","year":2015,"lang":"en","type":"article","venue":"2015 IEEE/ACM 37th IEEE International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Ontario Ministry of Research and Innovation; Natural Sciences and Engineering Research Council of Canada; University of Waterloo","keywords":"Computer science; Symbolic execution; Documentation; Heuristics; Software; Focus (optics); Programming language; Software bug; Data mining; Software engineering; Operating system","score_opus":0.10146070391989331,"score_gpt":0.33731170688533973,"score_spread":0.23585100296544642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4242156770","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025638312,0.0004904164,0.93158156,0.00032543676,0.00006410712,0.00023094579,0.0006650249,0.039185002,0.0018192078],"genre_scores_gemma":[0.13156514,0.0002451346,0.8634835,0.00016104922,0.000033469758,0.00022253212,0.0018197608,0.0013492474,0.0011201587],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99425584,0.0018197382,0.00049657485,0.00061049353,0.0025505223,0.00026677788],"domain_scores_gemma":[0.9831942,0.008926465,0.0015313207,0.0030436923,0.0030172695,0.00028711127],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002908853,0.0019331245,0.0010216925,0.0031923966,0.0004326653,0.001530168,0.0024279936,0.00090139563,0.00382605],"category_scores_gemma":[0.0170148,0.00058993534,0.00091724354,0.0019433616,0.001078095,0.0026330587,0.0019029794,0.0017040774,0.0012826566],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004816783,0.0005704671,0.00681827,0.00085164193,0.00015301522,0.00031937254,0.00045851304,0.07512581,0.058758583,0.011108315,0.013810289,0.8315441],"study_design_scores_gemma":[0.00033942773,0.00052105455,0.0017627047,0.0001244042,0.000101595615,0.00042796714,0.00013808874,0.8721235,0.09501457,0.011152543,0.018181067,0.000113045346],"about_ca_topic_score_codex":0.005052219,"about_ca_topic_score_gemma":0.007961065,"teacher_disagreement_score":0.005052219,"about_ca_system_score_codex":0.0009078718,"about_ca_system_score_gemma":0.0027728032,"threshold_uncertainty_score":0.015383661},"labels":[],"label_agreement":null},{"id":"W4242700588","doi":"10.1109/emsoft.2015.7318273","title":"A framework for mining hybrid automata from input/output traces","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Automaton; Abstraction; Inference; Cluster analysis; Theoretical computer science; Hybrid system; Domain (mathematical analysis); Automata theory; Data mining; Artificial intelligence; Machine learning","score_opus":0.09249891101697935,"score_gpt":0.319693611244333,"score_spread":0.22719470022735364,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4242700588","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001357018,0.000068732494,0.9964012,0.000046175053,0.0000044374174,0.000099443874,0.00027294137,0.0015815991,0.00016844866],"genre_scores_gemma":[0.03810112,0.00010364702,0.9599461,0.000033761477,0.000009868723,0.00026983113,0.0010589922,0.00015039719,0.00032632728],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974669,0.0005850736,0.00033836736,0.0007425233,0.0007472929,0.00011970605],"domain_scores_gemma":[0.9947463,0.0029902752,0.00052085443,0.00095121504,0.0006708843,0.000120471894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002792363,0.0018219105,0.0010724417,0.0046592476,0.001083389,0.0025506455,0.0028096188,0.0015576229,0.0020219109],"category_scores_gemma":[0.01276311,0.0011042841,0.0033818316,0.0026805536,0.0014998856,0.0030112693,0.001970465,0.0021182278,0.00074242224],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016774036,0.00040867465,0.009956973,0.0013061608,0.00054094556,0.0012907331,0.0014539117,0.43368384,0.019100344,0.11445853,0.0052154516,0.41241667],"study_design_scores_gemma":[0.00002925295,0.0000895481,0.0009144705,0.00011737964,0.00009256061,0.00050806097,0.00017290206,0.9100914,0.0071624676,0.07153562,0.009225679,0.000060816354],"about_ca_topic_score_codex":0.009655136,"about_ca_topic_score_gemma":0.013321808,"teacher_disagreement_score":0.009655136,"about_ca_system_score_codex":0.0010918857,"about_ca_system_score_gemma":0.0026803622,"threshold_uncertainty_score":0.019197881},"labels":[],"label_agreement":null},{"id":"W4242926184","doi":"10.1002/9780470612514","title":"Software Specification Methods","year":2006,"lang":"en","type":"book","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Software; Programming language; Software engineering","score_opus":0.03658288294643618,"score_gpt":0.32019772551635645,"score_spread":0.28361484256992026,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4242926184","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00096839305,0.013705761,0.60827094,0.0025866553,0.0021675425,0.00018752934,0.0008956441,0.004677904,0.3665396],"genre_scores_gemma":[0.010253387,0.013380909,0.12965976,0.0010440457,0.0011329425,0.00032140332,0.002879592,0.003181714,0.83814627],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99847275,0.00022516874,0.00010478422,0.00023518087,0.0009060025,0.000056087058],"domain_scores_gemma":[0.9984364,0.0006896868,0.00004213814,0.00025390316,0.0005158311,0.00006209241],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015444785,0.002749252,0.0015474041,0.0033503664,0.0008146367,0.0034896305,0.0017847592,0.0011002755,0.08357849],"category_scores_gemma":[0.0031161325,0.0013649175,0.0014448239,0.0043735616,0.0013828232,0.004362811,0.0013946826,0.0037425677,0.05486915],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000023432216,0.00005501409,0.00011392522,0.00057367835,0.000023030576,0.000071276874,0.00030388375,0.0017801275,0.0021219554,0.14770217,0.33117938,0.5160522],"study_design_scores_gemma":[0.000016852437,0.000024183968,0.00016707431,0.0002591508,0.000015826252,0.00022776061,0.000057581186,0.003880194,0.001172419,0.073032655,0.92112833,0.0000179278],"about_ca_topic_score_codex":0.001781607,"about_ca_topic_score_gemma":0.002861081,"teacher_disagreement_score":0.08357849,"about_ca_system_score_codex":0.0018748955,"about_ca_system_score_gemma":0.0014957172,"threshold_uncertainty_score":0.27959794},"labels":[],"label_agreement":null},{"id":"W4243225916","doi":"10.1007/978-1-4614-4325-4_6","title":"Applications","year":2012,"lang":"en","type":"book-chapter","venue":"CMS books in mathematics","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science","score_opus":0.04060258556210177,"score_gpt":0.27200067472071837,"score_spread":0.2313980891586166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4243225916","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011759901,0.0014420333,0.030370196,0.0015956565,0.0020887726,0.00021683911,0.004285023,0.010993998,0.94783145],"genre_scores_gemma":[0.0071528614,0.0012490947,0.012043909,0.0011215516,0.00046868896,0.00016977054,0.0064967033,0.0021647098,0.96913254],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99925154,0.000078315235,0.000029253606,0.00015201452,0.00041182205,0.00007697368],"domain_scores_gemma":[0.99914646,0.00011412632,0.000028436585,0.00023163218,0.00033846582,0.00014081811],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.000591376,0.0012491888,0.00058419147,0.0017131575,0.0011441882,0.0036626193,0.0017661756,0.001264697,0.5295854],"category_scores_gemma":[0.002480643,0.0003863139,0.00064746704,0.001736792,0.00035449755,0.0027720989,0.0028602486,0.0012536658,0.467425],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006433233,0.00006043394,0.00019227012,0.00024020043,0.0000087468725,0.00008081118,0.00008400358,0.0003248505,0.0020145415,0.022870531,0.6763184,0.29774088],"study_design_scores_gemma":[0.000007709372,0.000011946272,0.00012685702,0.00005077696,0.0000046757737,0.00007251314,0.000029599922,0.00038863768,0.00072162796,0.0061213523,0.99245775,0.0000066684574],"about_ca_topic_score_codex":0.0011827209,"about_ca_topic_score_gemma":0.0017799143,"teacher_disagreement_score":0.5295854,"about_ca_system_score_codex":0.00056710804,"about_ca_system_score_gemma":0.00091718364,"threshold_uncertainty_score":0.6709893},"labels":[],"label_agreement":null},{"id":"W4243258316","doi":"10.1109/icse.2004.1317431","title":"Using simulation to empirically investigate test coverage criteria based on statechart","year":2004,"lang":"en","type":"article","venue":"Proceedings. 26th International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Class (philosophy); Test case; Test (biology); Code coverage; Reliability engineering; Data mining; Machine learning; Artificial intelligence; Programming language; Software; Engineering","score_opus":0.08422974191687765,"score_gpt":0.34065714063343316,"score_spread":0.2564273987165555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4243258316","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91784364,0.00035413852,0.074991085,0.00028224988,0.000015999514,0.00035167343,0.0005312691,0.00017710128,0.0054528015],"genre_scores_gemma":[0.9839093,0.00008745629,0.015407517,0.000020495585,0.000003454226,0.00021524927,0.00022510285,0.000013140675,0.00011839439],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98638165,0.0098522,0.0007267259,0.00062461256,0.0018954478,0.0005193037],"domain_scores_gemma":[0.6313094,0.3485192,0.0070461095,0.0066753356,0.005915994,0.0005339339],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016257575,0.0009182334,0.00070639414,0.0032308104,0.0004142279,0.0015403961,0.001070898,0.0013753984,0.0017126804],"category_scores_gemma":[0.12115686,0.00046490273,0.0009742308,0.002759085,0.0010332429,0.00249911,0.0009287276,0.001057558,0.00014118594],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008440048,0.000649492,0.055247776,0.00027552887,0.00038335417,0.00011462183,0.00045106583,0.89943266,0.0020808436,0.018302131,0.0003996496,0.021818891],"study_design_scores_gemma":[0.0001021612,0.0012313308,0.010396491,0.00006859204,0.00013956444,0.00008623157,0.00026846724,0.9746907,0.0033019064,0.009061212,0.0006156492,0.000037724414],"about_ca_topic_score_codex":0.0048479,"about_ca_topic_score_gemma":0.0051911334,"teacher_disagreement_score":0.016257575,"about_ca_system_score_codex":0.0028175174,"about_ca_system_score_gemma":0.0013780219,"threshold_uncertainty_score":0.08597928},"labels":[],"label_agreement":null},{"id":"W4243372325","doi":"10.22215/etd/2016-11428","title":"Test Generation from an Extended Finite State Machine as a Multiobjective Optimization Problem","year":2016,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Test suite; Extended finite-state machine; Computer science; Finite-state machine; Test case; Test (biology); Model-based testing; Suite; Software; Genetic algorithm; Code coverage; Reliability engineering; Algorithm; Machine learning; Engineering; Programming language","score_opus":0.014348365568041761,"score_gpt":0.28408740907707475,"score_spread":0.269739043509033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4243372325","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06998398,0.00028571035,0.9260183,0.0003741364,0.000028069142,0.00012419862,0.00006960416,0.00026983858,0.0028460706],"genre_scores_gemma":[0.57861525,0.00029924663,0.4174946,0.00014926687,0.000028692026,0.00036434914,0.00023611121,0.00008729241,0.0027251747],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988965,0.0006044412,0.000055625376,0.00014803474,0.00020429962,0.000091040965],"domain_scores_gemma":[0.99701774,0.0024292981,0.00015886822,0.00009568748,0.0002409013,0.000057510446],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001745934,0.00081319606,0.00076365646,0.00085914537,0.00029409592,0.00086629426,0.00067123084,0.0011606202,0.001380358],"category_scores_gemma":[0.0038963621,0.00034534917,0.0011577273,0.00069140707,0.0008524217,0.0007137127,0.0006609468,0.0010573339,0.00013206716],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003640526,0.00004593445,0.00047677825,0.000050872008,0.000026764725,0.000093784765,0.000045181863,0.96982265,0.0017542627,0.0056996886,0.00023802782,0.021709546],"study_design_scores_gemma":[0.000012757375,0.00005295552,0.00015985631,0.0000131174265,0.0000124638245,0.00002865093,0.000013663217,0.9929304,0.00094297825,0.005433373,0.00039499183,0.0000048077995],"about_ca_topic_score_codex":0.002080138,"about_ca_topic_score_gemma":0.0017744893,"teacher_disagreement_score":0.002080138,"about_ca_system_score_codex":0.00087837374,"about_ca_system_score_gemma":0.0010465059,"threshold_uncertainty_score":0.009233534},"labels":[],"label_agreement":null},{"id":"W4243778460","doi":"10.1145/2775054.2694394","title":"Dual Execution for On the Fly Fine Grained Execution Comparison","year":2015,"lang":"en","type":"article","venue":"ACM SIGPLAN Notices","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Defense Advanced Research Projects Agency; National Science Foundation","keywords":"Computer science; Debugging; Programming language; Distributed computing","score_opus":0.13166431278598673,"score_gpt":0.32846661103871233,"score_spread":0.1968022982527256,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4243778460","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035826836,0.00031009523,0.9457643,0.00016576704,0.00010474146,0.00019718148,0.00025207511,0.013957174,0.0034217802],"genre_scores_gemma":[0.42042154,0.000112605674,0.5741423,0.00017262244,0.000038790797,0.0002867753,0.0005075591,0.0021511605,0.0021666505],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9948931,0.0014057185,0.0004969561,0.001053921,0.0016633887,0.00048693368],"domain_scores_gemma":[0.98965424,0.0037136527,0.0009292654,0.0042150966,0.0011543897,0.00033330885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034522521,0.00114013,0.00092662574,0.0023930056,0.0007454317,0.0017904993,0.0023667847,0.00091338315,0.004392032],"category_scores_gemma":[0.010810466,0.00072459993,0.0005883513,0.001644358,0.0016896084,0.004685779,0.0041061332,0.0020705813,0.0011943358],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003489928,0.0006096875,0.014756872,0.00074203516,0.00014182465,0.00068797776,0.0016907362,0.029619306,0.14904189,0.07498483,0.009890091,0.7143448],"study_design_scores_gemma":[0.00035984544,0.0011518997,0.006088391,0.00024382726,0.00015117678,0.0011348071,0.00040708532,0.5371127,0.28495562,0.10187128,0.06626252,0.0002609394],"about_ca_topic_score_codex":0.0014101644,"about_ca_topic_score_gemma":0.002000622,"teacher_disagreement_score":0.004392032,"about_ca_system_score_codex":0.0010026385,"about_ca_system_score_gemma":0.0014532005,"threshold_uncertainty_score":0.018257499},"labels":[],"label_agreement":null},{"id":"W4243891992","doi":"10.1109/icse.2013.6606688","title":"On extracting unit tests from interactive live programming sessions","year":2013,"lang":"en","type":"article","venue":"2013 35th International Conference on Software Engineering (ICSE)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Leverage (statistics); Session (web analytics); Unit testing; Cluster analysis; Software testing; Software; Software engineering; Machine learning; Programming language; World Wide Web","score_opus":0.042236101726807265,"score_gpt":0.29588421699843037,"score_spread":0.2536481152716231,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4243891992","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09250413,0.0003378472,0.89626455,0.0001665641,0.00004536575,0.00037401303,0.0011942565,0.006712435,0.0024007214],"genre_scores_gemma":[0.3354273,0.00042124745,0.65396994,0.00009482267,0.00006245427,0.0005257118,0.004680514,0.001329235,0.0034887043],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99787974,0.000578217,0.0001690033,0.00045326637,0.0007806425,0.00013919635],"domain_scores_gemma":[0.9773187,0.012808698,0.0022512083,0.004087348,0.003076507,0.00045744595],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012964513,0.0009993312,0.0007113195,0.0036917971,0.0005800851,0.001522921,0.0017744562,0.00103395,0.0026164611],"category_scores_gemma":[0.020738015,0.00041331572,0.00046666432,0.003227851,0.0008118447,0.0021691313,0.0012413923,0.0011853228,0.0014957236],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003676311,0.0002713019,0.015348364,0.00061222754,0.00006748667,0.0011198608,0.0034324501,0.011571474,0.072533436,0.0051244004,0.0032937508,0.8862577],"study_design_scores_gemma":[0.00013162063,0.0012506995,0.10709892,0.0007557246,0.00020310142,0.006366993,0.004949175,0.46091548,0.2996387,0.05331427,0.06492903,0.00044633486],"about_ca_topic_score_codex":0.0015685621,"about_ca_topic_score_gemma":0.0031261693,"teacher_disagreement_score":0.0036917971,"about_ca_system_score_codex":0.00038248856,"about_ca_system_score_gemma":0.0005596395,"threshold_uncertainty_score":0.008752942},"labels":[],"label_agreement":null},{"id":"W4246285000","doi":"10.32920/ryerson.14663895.v1","title":"Whole program analysis of Java programs for virtual calls and exception handling","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Call stack; Computer science; Java; Exception handling; Stack (abstract data type); Operating system; Static analysis; Virtual machine; strictfp; Code (set theory); Program analysis; Programming language; Real time Java","score_opus":0.04273522055698996,"score_gpt":0.3202934536791171,"score_spread":0.27755823312212713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246285000","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19216175,0.0003881581,0.7947842,0.000121362886,0.00003852491,0.000101194295,0.00026237292,0.009260761,0.0028817197],"genre_scores_gemma":[0.6554078,0.0003330451,0.33637312,0.00011502971,0.000038641883,0.00016189441,0.0007253488,0.003016585,0.003828494],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988181,0.00025925506,0.00007061623,0.00026400157,0.00044636088,0.00014166946],"domain_scores_gemma":[0.99618214,0.0020527295,0.00033261054,0.0007638411,0.0005910099,0.000077535246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008131248,0.0008526336,0.0006611261,0.0011322481,0.00067612214,0.0013647622,0.00092381804,0.00055231096,0.00309559],"category_scores_gemma":[0.0044954917,0.00037738617,0.0010512342,0.0010363987,0.00079472054,0.0017692525,0.00094271294,0.0008908759,0.0004765472],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010971565,0.00036698816,0.019245015,0.001063896,0.000316046,0.0011924108,0.0012837957,0.05426593,0.43447196,0.029271107,0.0044698524,0.45295584],"study_design_scores_gemma":[0.00011586878,0.0005867475,0.019486327,0.00014214775,0.00037349507,0.0011510121,0.00045949966,0.51067555,0.41339988,0.038193878,0.015288462,0.00012716356],"about_ca_topic_score_codex":0.00090410555,"about_ca_topic_score_gemma":0.0009605821,"teacher_disagreement_score":0.00309559,"about_ca_system_score_codex":0.00041393042,"about_ca_system_score_gemma":0.0012216148,"threshold_uncertainty_score":0.01035583},"labels":[],"label_agreement":null},{"id":"W4246412311","doi":"10.1002/jcd.21602","title":"Variable strength covering arrays","year":2018,"lang":"en","type":"article","venue":"Journal of Combinatorial Designs","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hypergraph; Mathematics; Variable (mathematics); Homomorphism; Combinatorics; Order (exchange); Discrete mathematics; Mathematical analysis","score_opus":0.02761422325529419,"score_gpt":0.26846090353492696,"score_spread":0.24084668027963277,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246412311","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17713095,0.00038405537,0.788734,0.0002833522,0.00013715719,0.000082401544,0.00039951364,0.0010976839,0.031750955],"genre_scores_gemma":[0.8158117,0.00035302615,0.17353235,0.00029465035,0.00012620242,0.00019203732,0.00050922576,0.0003703714,0.008810375],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99832326,0.00054646935,0.000103981074,0.00031076785,0.0004919531,0.00022352346],"domain_scores_gemma":[0.9944712,0.0025275312,0.0006223689,0.0012702937,0.00082478556,0.00028387122],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078443607,0.0005358878,0.00042339892,0.0012196033,0.0006014818,0.0015435632,0.00084100757,0.0006168619,0.005432137],"category_scores_gemma":[0.0041156965,0.0004395774,0.00052263215,0.0015626015,0.001094435,0.0023388837,0.0014846284,0.00086917856,0.001216023],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023777425,0.00007903734,0.0039364584,0.00022580862,0.00006802622,0.00066216924,0.00043797112,0.042473003,0.0450778,0.7686029,0.0044893087,0.13370973],"study_design_scores_gemma":[0.0000524569,0.00037574605,0.0019188514,0.00011120833,0.00009507128,0.0020302027,0.00036530948,0.1699622,0.066002294,0.71031755,0.048675347,0.00009379428],"about_ca_topic_score_codex":0.00020957952,"about_ca_topic_score_gemma":0.00019367319,"teacher_disagreement_score":0.005432137,"about_ca_system_score_codex":0.000439392,"about_ca_system_score_gemma":0.00030554636,"threshold_uncertainty_score":0.018172264},"labels":[],"label_agreement":null},{"id":"W4246883339","doi":"10.22215/etd/2014-10588","title":"Mapping ACL to JavaMOP: A Feasibility Study","year":2014,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"USable; Suite; Computer science; Software engineering; Test suite; Debugging; Java; Test case; Programming language; Reliability engineering; Engineering; Machine learning; World Wide Web","score_opus":0.04234093436177168,"score_gpt":0.33030948437993707,"score_spread":0.2879685500181654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246883339","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.937906,0.00022531959,0.023370717,0.0015951722,0.00017754029,0.0037801338,0.0011372578,0.0014814443,0.030326296],"genre_scores_gemma":[0.94013864,0.00019618419,0.05212307,0.00032987347,0.0000240205,0.0011016249,0.0014950613,0.00015829208,0.0044332533],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942842,0.0030819832,0.00034437992,0.0004975053,0.0011907854,0.0006012236],"domain_scores_gemma":[0.9672417,0.020670924,0.0006278661,0.0032095434,0.006351288,0.0018986723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01055906,0.0006972931,0.00042703975,0.00094476796,0.00078067207,0.002081644,0.00320264,0.0017726338,0.013931439],"category_scores_gemma":[0.02938226,0.00036799698,0.0006138222,0.0007966703,0.0007785469,0.0041754735,0.0012475972,0.00129881,0.0027075512],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.025076434,0.054493383,0.05787954,0.0054247826,0.00027951854,0.0049102134,0.00501791,0.14221388,0.07366057,0.034261312,0.025562752,0.5712197],"study_design_scores_gemma":[0.009511412,0.06135089,0.034258213,0.00077006774,0.0005229897,0.0017631515,0.013203618,0.71708053,0.085508026,0.018830441,0.05690985,0.00029085175],"about_ca_topic_score_codex":0.011108836,"about_ca_topic_score_gemma":0.00641685,"teacher_disagreement_score":0.013931439,"about_ca_system_score_codex":0.001650443,"about_ca_system_score_gemma":0.0033982624,"threshold_uncertainty_score":0.05584234},"labels":[],"label_agreement":null},{"id":"W4246889973","doi":"10.22215/etd/2005-08236","title":"Technique and automation for testing of commercial-off-the-shelf components","year":2005,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage; Library and Archives Canada","funders":"","keywords":"Computer science; Automation; Engineering; Mechanical engineering","score_opus":0.039032232194455584,"score_gpt":0.30262781846636155,"score_spread":0.263595586271906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246889973","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013540563,0.00030813267,0.9769654,0.00015533417,0.000069818154,0.00011182017,0.00006504256,0.004698682,0.004085331],"genre_scores_gemma":[0.1976883,0.00049808447,0.79385936,0.00015513509,0.00004317471,0.00020935919,0.00039009738,0.0005822548,0.0065742102],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9971687,0.00055234635,0.00018456928,0.00034424008,0.0016020593,0.00014800478],"domain_scores_gemma":[0.99627304,0.0013589542,0.00027750546,0.0012494719,0.00078489026,0.00005611367],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00093076914,0.0007613683,0.00041353778,0.0014620849,0.0003894154,0.00090471265,0.0012477499,0.0008102248,0.003141698],"category_scores_gemma":[0.0049789706,0.00047400768,0.000517269,0.00084698247,0.0007976362,0.001061974,0.0008288342,0.0012906307,0.0020011752],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001321102,0.0001899767,0.0019616992,0.0004564753,0.000048497008,0.0006242125,0.0006949284,0.003890691,0.37313977,0.019391768,0.00580959,0.59366035],"study_design_scores_gemma":[0.00012254335,0.00074794004,0.0066945306,0.0002534829,0.00013941433,0.007972797,0.00027644343,0.101557404,0.7538735,0.023957103,0.10431298,0.00009175876],"about_ca_topic_score_codex":0.000782857,"about_ca_topic_score_gemma":0.0011271376,"teacher_disagreement_score":0.003141698,"about_ca_system_score_codex":0.0002542199,"about_ca_system_score_gemma":0.00064304454,"threshold_uncertainty_score":0.010510027},"labels":[],"label_agreement":null},{"id":"W4246930860","doi":"10.22215/etd/2007-08494","title":"A UML profile for developing airworthiness-compliant (RTCA DO-178B) safety-critical software","year":2007,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage","funders":"","keywords":"Airworthiness; Engineering; Computer science; Aeronautics; Political science; Certification","score_opus":0.032402321004011324,"score_gpt":0.36067874967522084,"score_spread":0.32827642867120954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246930860","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008964243,0.0005358448,0.94082505,0.001087859,0.00014541656,0.0030534195,0.0033780036,0.023115885,0.018894233],"genre_scores_gemma":[0.035010356,0.0011954133,0.93225724,0.00049738795,0.00004246365,0.0023850137,0.0105038425,0.00277937,0.015328973],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99571085,0.0012733735,0.00080965256,0.00022005438,0.0017426729,0.00024330495],"domain_scores_gemma":[0.99277437,0.0018982714,0.00095672265,0.0014817726,0.0023611486,0.0005276397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064205388,0.0016350019,0.0006818271,0.0035108028,0.00081781275,0.0030361966,0.0021921229,0.0020411536,0.0042439783],"category_scores_gemma":[0.013528534,0.0013867287,0.0008939996,0.0014866756,0.00064707873,0.0023882089,0.0015073825,0.0023683086,0.008835007],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004594827,0.0011005689,0.0063996348,0.002025008,0.00008011537,0.0020676276,0.0044402974,0.035649534,0.11611813,0.10534316,0.11069801,0.61561835],"study_design_scores_gemma":[0.00020553115,0.00037015087,0.0041735684,0.002706294,0.000082028695,0.003441453,0.0006540758,0.09555962,0.05154602,0.029048756,0.8119948,0.00021767478],"about_ca_topic_score_codex":0.0055960244,"about_ca_topic_score_gemma":0.009908821,"teacher_disagreement_score":0.0064205388,"about_ca_system_score_codex":0.0015278951,"about_ca_system_score_gemma":0.0060309432,"threshold_uncertainty_score":0.033955455},"labels":[],"label_agreement":null},{"id":"W4247047556","doi":"10.1145/2345156.2254091","title":"Parallelizing top-down interprocedural analyses","year":2012,"lang":"en","type":"article","venue":"ACM SIGPLAN Notices","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Reachability; Scalability; Call graph; Graph; Modular design; Context (archaeology); Parallelism (grammar); Programming language; Top-down and bottom-up design; Static analysis; Theoretical computer science; Parallel computing; Database","score_opus":0.08809835399604228,"score_gpt":0.36376495046101753,"score_spread":0.27566659646497527,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4247047556","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03134104,0.00042982106,0.9475266,0.00055239873,0.00012058774,0.00031123945,0.0003410778,0.014856832,0.004520402],"genre_scores_gemma":[0.30810735,0.0003557739,0.68013936,0.00057521707,0.00017292895,0.00043610495,0.0013956869,0.003861637,0.004956041],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9934608,0.0013146086,0.00033586967,0.0010545442,0.002816964,0.001017257],"domain_scores_gemma":[0.9907244,0.003846853,0.0005206686,0.0032131854,0.0014033819,0.00029143508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032060565,0.0023798428,0.0015223701,0.002402756,0.0012781688,0.0029966643,0.0031107129,0.00094795856,0.0038926883],"category_scores_gemma":[0.011975425,0.0011123467,0.0034481206,0.0015790015,0.0024599512,0.0047432543,0.0061230017,0.0028761649,0.0017080699],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011262324,0.000540707,0.010689345,0.001133961,0.00042934675,0.0010376147,0.0018109924,0.23171407,0.075133845,0.09787559,0.015649978,0.56285834],"study_design_scores_gemma":[0.00013688143,0.00022284595,0.0019136869,0.0001192366,0.00025118358,0.0002587848,0.0002876385,0.7123999,0.063416585,0.20176063,0.01911439,0.00011817303],"about_ca_topic_score_codex":0.0082319705,"about_ca_topic_score_gemma":0.010456862,"teacher_disagreement_score":0.0082319705,"about_ca_system_score_codex":0.0020094095,"about_ca_system_score_gemma":0.004686271,"threshold_uncertainty_score":0.016955435},"labels":[],"label_agreement":null},{"id":"W4247094842","doi":"10.1109/icse.2013.6606563","title":"Comparing Multi-Point Stride Coverage and dataflow coverage","year":2013,"lang":"en","type":"article","venue":"2013 35th International Conference on Software Engineering (ICSE)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Ontario","keywords":"Computer science; Dataflow; Overhead (engineering); Tuple; Point (geometry); Parallel computing; Algorithm; Programming language; Mathematics","score_opus":0.04861954004122281,"score_gpt":0.26537905002441176,"score_spread":0.21675950998318894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4247094842","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.71378416,0.0018323835,0.27342993,0.00021658055,0.000037741334,0.00015095527,0.0020255228,0.0028065774,0.0057160477],"genre_scores_gemma":[0.9598868,0.00020984875,0.037703533,0.000036015488,0.000022354403,0.00009134145,0.0014458196,0.00021749576,0.0003867659],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9921564,0.0015734042,0.0005505753,0.00097328774,0.004147743,0.00059857615],"domain_scores_gemma":[0.9112913,0.0676157,0.0056475587,0.008056295,0.00624405,0.001144986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052548195,0.00091606757,0.0010521731,0.008009197,0.00040204194,0.0011336186,0.0013806446,0.0012169089,0.0017771748],"category_scores_gemma":[0.047466584,0.00037538187,0.0008951707,0.00473882,0.0011577597,0.003595809,0.00197485,0.000773675,0.00027865992],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016358499,0.00036257974,0.24667884,0.0007113582,0.00051097793,0.00078629376,0.000673971,0.4134214,0.026779225,0.014985892,0.0026503082,0.2908033],"study_design_scores_gemma":[0.000059790033,0.00091690733,0.06275834,0.00009172252,0.00011344922,0.0010371063,0.00025342073,0.89157796,0.029119933,0.011426838,0.0025663376,0.00007824059],"about_ca_topic_score_codex":0.0021120282,"about_ca_topic_score_gemma":0.002095477,"teacher_disagreement_score":0.008009197,"about_ca_system_score_codex":0.0006700027,"about_ca_system_score_gemma":0.0005514333,"threshold_uncertainty_score":0.027790487},"labels":[],"label_agreement":null},{"id":"W4247417056","doi":"10.1145/2070337.2070357","title":"Enhancing spark's contract checking facilities using symbolic execution","year":2011,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Symbolic execution; SPARK (programming language); Software engineering; Programming language; Usability; Automation; Model checking; Design by contract; Software; Formal methods; Software development; Software construction; Operating system; Engineering","score_opus":0.08202945841302203,"score_gpt":0.27001853384797886,"score_spread":0.18798907543495683,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4247417056","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021896785,0.00012174019,0.93833774,0.00025983356,0.00007073858,0.00010803609,0.00016483547,0.034614075,0.004426179],"genre_scores_gemma":[0.27274826,0.00023263214,0.7174639,0.00019372632,0.0000563508,0.00014347969,0.0006698678,0.0047410224,0.0037507294],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99292094,0.0017947213,0.0004959091,0.00080663594,0.003465971,0.00051580457],"domain_scores_gemma":[0.9769852,0.013000907,0.0013683619,0.00529558,0.0029572973,0.0003926664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060793627,0.0011085701,0.0009524526,0.0019106342,0.00086605526,0.0019852188,0.0028211984,0.0010211004,0.0043952297],"category_scores_gemma":[0.021011436,0.0010097225,0.0014857238,0.0011323747,0.0028121595,0.0037661472,0.003160264,0.002455817,0.0014411164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016553026,0.00061417377,0.012189195,0.0010795539,0.0002931161,0.0014517652,0.0021058524,0.16269131,0.07781514,0.18744126,0.02437965,0.52828383],"study_design_scores_gemma":[0.00041959548,0.00030359573,0.0011083017,0.00013908841,0.00011434415,0.000854091,0.00015620304,0.8128307,0.08758002,0.05303467,0.043298468,0.00016086877],"about_ca_topic_score_codex":0.005567498,"about_ca_topic_score_gemma":0.0055385963,"teacher_disagreement_score":0.0060793627,"about_ca_system_score_codex":0.0011228166,"about_ca_system_score_gemma":0.004182087,"threshold_uncertainty_score":0.032151103},"labels":[],"label_agreement":null},{"id":"W4248114548","doi":"10.1109/sbst.2015.20","title":"JTExpert at the Third Unit Testing Tool Competition","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Unit testing; Test suite; Computer science; Java; Suite; CONTEST; Ranking (information retrieval); Source code; Software testing; Programming language; Code coverage; Code (set theory); Software quality; Software; Competition (biology); Test case; Software engineering; Artificial intelligence; Machine learning; Software development; Set (abstract data type)","score_opus":0.09944752895663983,"score_gpt":0.2917210306793673,"score_spread":0.19227350172272745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248114548","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6817721,0.008483976,0.09149721,0.016736906,0.016248181,0.0016427336,0.016582789,0.035309926,0.1317261],"genre_scores_gemma":[0.7876282,0.0009760552,0.050772686,0.002464411,0.001326687,0.0007782704,0.05804664,0.004550708,0.09345626],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.980352,0.004712207,0.0007071324,0.0019183003,0.008640625,0.0036697062],"domain_scores_gemma":[0.951263,0.008294771,0.00081640977,0.0025799738,0.020007918,0.017037963],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019721616,0.002409807,0.0021564285,0.005302225,0.0022299318,0.0053665964,0.0023295172,0.0032349974,0.018724246],"category_scores_gemma":[0.029339664,0.00047698306,0.0021072598,0.0029904,0.0011423609,0.003176751,0.004239592,0.0034959854,0.010787015],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025953185,0.0027228238,0.017717263,0.0007878072,0.00044994408,0.0011289202,0.0013208017,0.0076864013,0.012832201,0.006420413,0.55902517,0.38731283],"study_design_scores_gemma":[0.00260743,0.011164633,0.13186677,0.00068767584,0.00045818658,0.002953239,0.0029697753,0.09614392,0.039896373,0.014624043,0.6958694,0.0007585118],"about_ca_topic_score_codex":0.008428712,"about_ca_topic_score_gemma":0.017862713,"teacher_disagreement_score":0.019721616,"about_ca_system_score_codex":0.0028876804,"about_ca_system_score_gemma":0.0031864597,"threshold_uncertainty_score":0.10429913},"labels":[],"label_agreement":null},{"id":"W4248354362","doi":"10.22215/etd/2018-13273","title":"Comparison of Approaches to Category Partition Specifications, Selection Criteria, and the Impact of the ‘Error’ and ‘Single’ Annotations using Industrial Case Studies","year":2018,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Partition (number theory); Computer science; Selection (genetic algorithm); Code (set theory); Reliability engineering; Regression testing; Code coverage; Test (biology); White-box testing; Software; Data mining; Machine learning; Programming language; Engineering; Mathematics; Software system; Software construction","score_opus":0.5374655900029118,"score_gpt":0.4350745023918684,"score_spread":0.10239108761104343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248354362","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.83915406,0.0032094775,0.14658396,0.00083971064,0.00006956545,0.0008913541,0.00033059885,0.0010847322,0.007836588],"genre_scores_gemma":[0.7574703,0.00094525254,0.2390773,0.00016385855,0.000023386336,0.0004630733,0.00083137216,0.00028776823,0.00073761004],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.96621054,0.020866683,0.002330673,0.0017931947,0.00778398,0.0010149528],"domain_scores_gemma":[0.7394629,0.23241396,0.0073283315,0.007148688,0.011901886,0.0017441945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027965285,0.001695808,0.0010809375,0.0060687605,0.001010711,0.0031923011,0.0026270684,0.0019700401,0.0012452513],"category_scores_gemma":[0.10000811,0.0007288081,0.0012424771,0.0040086214,0.0015675118,0.0032081555,0.0023799287,0.0015714937,0.000208239],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036811817,0.0029023204,0.041505985,0.00312137,0.0008036304,0.0004974652,0.004396905,0.28854272,0.017842818,0.017733391,0.0030026452,0.61596954],"study_design_scores_gemma":[0.0008142521,0.0039620143,0.028926339,0.0008991769,0.0010890724,0.00070638687,0.0066655395,0.9039158,0.032245956,0.012895559,0.007596812,0.00028309296],"about_ca_topic_score_codex":0.006303095,"about_ca_topic_score_gemma":0.011671319,"teacher_disagreement_score":0.027965285,"about_ca_system_score_codex":0.0031182247,"about_ca_system_score_gemma":0.00356449,"threshold_uncertainty_score":0.14789635},"labels":[],"label_agreement":null},{"id":"W4248501463","doi":"10.1002/spe.839","title":"Oto, a generic and extensible tool for marking programming assignments","year":2007,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Université du Québec à Montréal","keywords":"Computer science; Extensibility; Workload; Task (project management); Programming language; Automation; Process (computing); Software engineering; Source code; Code (set theory); Operating system; Engineering; Systems engineering","score_opus":0.028268078525108157,"score_gpt":0.32067617705254514,"score_spread":0.292408098527437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248501463","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016965512,0.00022893731,0.8483173,0.00021111016,0.00025586024,0.0010396406,0.0014843297,0.1247445,0.006752818],"genre_scores_gemma":[0.10959487,0.0005623594,0.8510106,0.00028257363,0.00015533928,0.0013329141,0.0065647704,0.016010223,0.014486388],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9967894,0.00059673924,0.000625644,0.00049013336,0.0012417907,0.00025631825],"domain_scores_gemma":[0.9866519,0.006301859,0.0015565845,0.0028116275,0.0015298541,0.0011482548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038036972,0.0016987232,0.0010854081,0.003966409,0.000683181,0.0026700317,0.0029226507,0.0012541274,0.014624656],"category_scores_gemma":[0.024133924,0.0011285299,0.0009022567,0.0021019022,0.0010154558,0.003903323,0.0036398652,0.0014729019,0.0045234733],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011217903,0.0007291697,0.0059064804,0.0014905066,0.00008183363,0.0015700524,0.0018569187,0.008382333,0.03824449,0.013462896,0.053350642,0.87380296],"study_design_scores_gemma":[0.001410539,0.0014496592,0.01615257,0.00175217,0.00028987156,0.006858051,0.0010575496,0.19091573,0.11013869,0.048338637,0.62073314,0.0009034106],"about_ca_topic_score_codex":0.0011897604,"about_ca_topic_score_gemma":0.0010782342,"teacher_disagreement_score":0.014624656,"about_ca_system_score_codex":0.0006017946,"about_ca_system_score_gemma":0.0018126899,"threshold_uncertainty_score":0.048924327},"labels":[],"label_agreement":null},{"id":"W4248647026","doi":"10.1002/stvr.410","title":"Improving the coverage criteria of UML state machines using data flow analysis","year":2009,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Unified Modeling Language; Guard (computer science); Finite-state machine; Data mining; Data flow diagram; Test suite; Context (archaeology); State (computer science); Control flow; Data-flow analysis; Tree (set theory); Test case; Algorithm; Machine learning; Database; Programming language; Mathematics; Software","score_opus":0.04962212616320409,"score_gpt":0.3113156837427992,"score_spread":0.2616935575795951,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248647026","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34594256,0.00028792542,0.64980954,0.00026482882,0.000009750065,0.00022709157,0.00026866587,0.0015208675,0.0016687914],"genre_scores_gemma":[0.84857213,0.00008268785,0.15042144,0.000038452876,0.000011474681,0.00019696762,0.00037591715,0.00009891789,0.00020196383],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.99191767,0.004424823,0.00046736933,0.0004991555,0.0022930868,0.00039778653],"domain_scores_gemma":[0.9326225,0.0572468,0.0034577195,0.001940481,0.004344654,0.00038782996],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067224316,0.0010055059,0.00086887006,0.0067187496,0.000487898,0.0015717188,0.00087843405,0.00090583577,0.00093094737],"category_scores_gemma":[0.041230783,0.00046783884,0.001149456,0.0015596242,0.0011434419,0.0021754527,0.0013211973,0.0006263652,0.00013116981],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008056375,0.0004528644,0.048314147,0.0005080661,0.00018972535,0.00061050785,0.0010754016,0.5931373,0.03975538,0.030341052,0.001195753,0.2836141],"study_design_scores_gemma":[0.000031302025,0.00014085112,0.0025002747,0.000048215752,0.000035640533,0.00007986661,0.00006091521,0.9727895,0.015765525,0.007932138,0.0005965186,0.00001927684],"about_ca_topic_score_codex":0.0042762244,"about_ca_topic_score_gemma":0.0027272059,"teacher_disagreement_score":0.0067224316,"about_ca_system_score_codex":0.0016750196,"about_ca_system_score_gemma":0.001255769,"threshold_uncertainty_score":0.035552084},"labels":[],"label_agreement":null},{"id":"W4249894557","doi":"10.22215/etd/2010-08791","title":"On the round trip path strategy for state based testing","year":2010,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage; Library and Archives Canada","funders":"","keywords":"Path (computing); Computer science; Humanities; Political science; Art; Operating system","score_opus":0.06221995767341325,"score_gpt":0.3088627045222606,"score_spread":0.24664274684884735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4249894557","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014720527,0.00018477856,0.9687964,0.0003109967,0.000056092616,0.00019490131,0.00010775592,0.0023663568,0.01326233],"genre_scores_gemma":[0.53456664,0.00025728685,0.45206767,0.00039579964,0.000059993323,0.00030902875,0.0004144236,0.00067766604,0.011251616],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975873,0.0008995344,0.00013711788,0.00035071693,0.0007882955,0.00023699619],"domain_scores_gemma":[0.9940382,0.0038352374,0.00026442966,0.0010285575,0.00064430217,0.00018922787],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001882813,0.0009235145,0.000655429,0.0015483497,0.00052281824,0.00169611,0.0019215698,0.0009690523,0.010960807],"category_scores_gemma":[0.008311313,0.00041805158,0.00061520125,0.0009068565,0.0016654585,0.003301705,0.0017487321,0.0012484037,0.0022334983],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00103977,0.0002721399,0.0014534539,0.0002602894,0.00008638334,0.0004256755,0.00048400037,0.086040445,0.017819026,0.29209575,0.01021898,0.589804],"study_design_scores_gemma":[0.00014305563,0.00051343517,0.00049492036,0.00008777979,0.00009150322,0.00031892292,0.00013864263,0.74969065,0.014843944,0.22399172,0.009626561,0.000058890815],"about_ca_topic_score_codex":0.0032615766,"about_ca_topic_score_gemma":0.0036379304,"teacher_disagreement_score":0.010960807,"about_ca_system_score_codex":0.0008588033,"about_ca_system_score_gemma":0.0012334108,"threshold_uncertainty_score":0.036667585},"labels":[],"label_agreement":null},{"id":"W4250487516","doi":"10.1145/566172.566183","title":"Investigating the use of analysis contracts to support fault isolation in object oriented code","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Isolation (microbiology); Object-oriented programming; Fault detection and isolation; Software engineering; Programming language; Artificial intelligence","score_opus":0.08638116512189156,"score_gpt":0.29273032185097225,"score_spread":0.2063491567290807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4250487516","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8563218,0.00028676196,0.13643128,0.0008892672,0.000033412794,0.00012719524,0.00003828429,0.000522132,0.005349885],"genre_scores_gemma":[0.95369893,0.000074112664,0.045237616,0.0000690609,0.000007652577,0.000031349988,0.000027842283,0.00006224151,0.00079119165],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9945381,0.0030695822,0.00017523032,0.00026033062,0.0014295767,0.0005271479],"domain_scores_gemma":[0.9151572,0.068009086,0.0050194846,0.0060265297,0.005012988,0.00077473774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00785409,0.00038504563,0.00030975885,0.000657543,0.0008483067,0.00110321,0.0014121628,0.0013720257,0.0012306888],"category_scores_gemma":[0.05576396,0.00042294007,0.0003011212,0.0008093459,0.0014345879,0.0033425495,0.0010576511,0.0016150291,0.00013640654],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030562817,0.003926009,0.11112999,0.00072633487,0.00023626654,0.0013315487,0.009597794,0.16362682,0.08431544,0.15875532,0.0035152317,0.459783],"study_design_scores_gemma":[0.00025752455,0.0010544386,0.013300213,0.00008936765,0.00020650786,0.00043584584,0.0019078902,0.896627,0.04564218,0.03611932,0.0043032323,0.00005639261],"about_ca_topic_score_codex":0.0046735574,"about_ca_topic_score_gemma":0.004319845,"teacher_disagreement_score":0.00785409,"about_ca_system_score_codex":0.00066533225,"about_ca_system_score_gemma":0.0017951825,"threshold_uncertainty_score":0.041536868},"labels":[],"label_agreement":null},{"id":"W4251131822","doi":"10.1145/2786763.2694394","title":"Dual Execution for On the Fly Fine Grained Execution Comparison","year":2015,"lang":"en","type":"article","venue":"ACM SIGARCH Computer Architecture News","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Defense Advanced Research Projects Agency; National Science Foundation","keywords":"Computer science; Debugging; Reuse; Identification (biology); Distributed computing; Programming language","score_opus":0.07261865442083672,"score_gpt":0.30802820684058124,"score_spread":0.23540955241974454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251131822","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037779357,0.00032812404,0.943661,0.00017878629,0.0001137634,0.00019502526,0.00024309554,0.013877989,0.0036229682],"genre_scores_gemma":[0.431914,0.00011513896,0.5626967,0.00018364619,0.00004003445,0.00027767563,0.00049253815,0.002071629,0.002208624],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9950806,0.0013443602,0.00047188532,0.0010036096,0.0016308333,0.0004686817],"domain_scores_gemma":[0.99005985,0.003500711,0.0009068855,0.004062234,0.0011512423,0.00031910633],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033752024,0.0011116771,0.00091191614,0.0023930692,0.0007535406,0.0017817544,0.0023186577,0.0009018866,0.0042518484],"category_scores_gemma":[0.010511345,0.0006967889,0.0005666105,0.0016392978,0.0016559141,0.004685735,0.0039111865,0.0020537945,0.0011768315],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033716029,0.00061600516,0.014664345,0.0006987929,0.00014149105,0.00068340887,0.0016084535,0.02831198,0.15269615,0.07603932,0.010163053,0.7110054],"study_design_scores_gemma":[0.0003499329,0.0011352643,0.00596244,0.00023477506,0.00014367972,0.0011398931,0.0004001901,0.539164,0.28655666,0.1012062,0.06345025,0.00025674255],"about_ca_topic_score_codex":0.0013935528,"about_ca_topic_score_gemma":0.0019593774,"teacher_disagreement_score":0.0042518484,"about_ca_system_score_codex":0.0010139357,"about_ca_system_score_gemma":0.0014515677,"threshold_uncertainty_score":0.017849982},"labels":[],"label_agreement":null},{"id":"W4251245562","doi":"10.1002/stvr.396","title":"Transition covering tests for systems with queues","year":2008,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal; Nortel (Canada)","funders":"","keywords":"Context (archaeology); Computer science; Queue; Metric (unit); Concurrency; Cover (algebra); Model-based testing; Test (biology); Test case; Reliability engineering; Distributed computing; Programming language; Engineering; Operations management; Mechanical engineering","score_opus":0.0432607937311662,"score_gpt":0.25965007470477164,"score_spread":0.21638928097360544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251245562","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10449646,0.00022594816,0.890035,0.00015591813,0.000040206174,0.0001018095,0.00008203561,0.0020711687,0.0027915246],"genre_scores_gemma":[0.82992953,0.00012014106,0.16847193,0.00008785753,0.000043742213,0.00013157444,0.00018648585,0.00019845032,0.0008304048],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9960245,0.0013201302,0.0002981786,0.0005036467,0.0014506435,0.00040293924],"domain_scores_gemma":[0.9833786,0.013357203,0.0008699974,0.0010904087,0.0009893054,0.00031453677],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018609673,0.00060277455,0.0007121117,0.0014664794,0.0005490912,0.001397303,0.00094878714,0.0009506403,0.0017572214],"category_scores_gemma":[0.013760338,0.00039174236,0.0010133635,0.0008416457,0.0018284294,0.002040806,0.0019134635,0.0011909498,0.00021952765],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012134722,0.00029851368,0.01318149,0.00062985695,0.0002096593,0.0037373581,0.0023626315,0.35861984,0.069423415,0.29398885,0.0025851484,0.25374973],"study_design_scores_gemma":[0.00006892056,0.00029037485,0.0016723829,0.00012695136,0.000067151275,0.00093020767,0.00015509583,0.81108135,0.05254556,0.1258835,0.007112031,0.00006647371],"about_ca_topic_score_codex":0.0015122182,"about_ca_topic_score_gemma":0.0007698928,"teacher_disagreement_score":0.0018609673,"about_ca_system_score_codex":0.00072306825,"about_ca_system_score_gemma":0.0008125063,"threshold_uncertainty_score":0.0098418},"labels":[],"label_agreement":null},{"id":"W4252446732","doi":"10.1145/949415.949425","title":"*J","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Compiler; Java; Programming language; Just-in-time compilation; Metric (unit); Software engineering","score_opus":0.01687873843337306,"score_gpt":0.24021438045611965,"score_spread":0.22333564202274658,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4252446732","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014338388,0.0008463621,0.42452043,0.0047488683,0.003985535,0.0006907121,0.010795342,0.034909785,0.50516456],"genre_scores_gemma":[0.11063924,0.0013574837,0.39673364,0.0028733602,0.0011291391,0.0010715225,0.020805651,0.010523153,0.45486695],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984409,0.00020796714,0.00016580838,0.00036214845,0.0006467222,0.00017637321],"domain_scores_gemma":[0.9968598,0.00032956264,0.00020858747,0.0011140957,0.0011750701,0.00031290777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012162124,0.0006409996,0.0005105172,0.0018481964,0.0019394803,0.0035887484,0.0014069846,0.0008837458,0.12759238],"category_scores_gemma":[0.0056598545,0.0004162226,0.00062317296,0.0014889828,0.0006784581,0.0045092274,0.002969359,0.0013174709,0.06966392],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019100163,0.0001829968,0.00434394,0.00028131326,0.000035127476,0.00020272711,0.00065786677,0.0008606873,0.004986371,0.16205914,0.36564818,0.46055067],"study_design_scores_gemma":[0.000015771206,0.00005373884,0.0017467442,0.00005907193,0.000020162293,0.0003367514,0.00014906544,0.0034495608,0.0040377905,0.03341268,0.95667297,0.000045720655],"about_ca_topic_score_codex":0.0029505938,"about_ca_topic_score_gemma":0.0052156383,"teacher_disagreement_score":0.12759238,"about_ca_system_score_codex":0.0005834197,"about_ca_system_score_gemma":0.0017824662,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4253500466","doi":"10.22215/etd/2012-06901","title":"Improving testability with stateless method extraction","year":2012,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage; Library and Archives Canada","funders":"","keywords":"Testability; Stateless protocol; Computer science; Mathematics; Algorithm; Statistics","score_opus":0.01902002175544895,"score_gpt":0.32283269567276585,"score_spread":0.3038126739173169,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4253500466","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034405176,0.00065056933,0.92871666,0.0005253783,0.0001378833,0.00023069976,0.0006330274,0.031043174,0.0036574954],"genre_scores_gemma":[0.39864954,0.0006371805,0.58510894,0.00045068693,0.00009629423,0.00031114309,0.0033961027,0.004552708,0.0067974315],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9914443,0.0023954632,0.0010066014,0.0010862782,0.0035665822,0.0005008048],"domain_scores_gemma":[0.9564422,0.024053466,0.0023488028,0.01243523,0.004414808,0.00030551842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031179045,0.0016244225,0.0012916372,0.003243675,0.00058713485,0.003095459,0.0029861561,0.0014486995,0.008039414],"category_scores_gemma":[0.027895253,0.0013529699,0.0022120064,0.0021584323,0.001403141,0.0077716103,0.0027237467,0.0022301716,0.0024873244],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006137564,0.00041240721,0.004333415,0.0008664231,0.00016075604,0.0003469907,0.0003525619,0.028650226,0.061351493,0.023210611,0.008667996,0.8710333],"study_design_scores_gemma":[0.00023588023,0.00048553312,0.0024362002,0.00026982976,0.0002873468,0.0006943148,0.00013167846,0.67399514,0.21653308,0.082270466,0.022519898,0.00014059512],"about_ca_topic_score_codex":0.002282108,"about_ca_topic_score_gemma":0.004011434,"teacher_disagreement_score":0.008039414,"about_ca_system_score_codex":0.001188357,"about_ca_system_score_gemma":0.0026709817,"threshold_uncertainty_score":0.02689451},"labels":[],"label_agreement":null},{"id":"W4254152538","doi":"10.1109/ast.2015.13","title":"Adaptive Random Testing by Static Partitioning","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Random testing; Computer science; Overhead (engineering); Point (geometry); Isolation (microbiology); Test case; Parallel computing; Algorithm; Distributed computing; Theoretical computer science; Machine learning; Mathematics","score_opus":0.0889890986322632,"score_gpt":0.27574200080274397,"score_spread":0.18675290217048077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254152538","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021287689,0.00019275167,0.97417325,0.000115392155,0.000039536062,0.0000978596,0.00003305445,0.0013135392,0.002746863],"genre_scores_gemma":[0.5464041,0.00019356173,0.44935548,0.00019271189,0.00004958742,0.00026498974,0.00020022818,0.00044482702,0.0028944616],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976254,0.00079602917,0.00012536866,0.00040774632,0.00081230904,0.00023316422],"domain_scores_gemma":[0.9944998,0.0025854318,0.0004530716,0.0012319274,0.001071455,0.00015824033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014729005,0.0010223716,0.0007935151,0.001405205,0.00061978435,0.00088849885,0.0020190494,0.0007127513,0.0024937452],"category_scores_gemma":[0.008716422,0.0004779026,0.0007109339,0.0008680517,0.0010381038,0.0019909372,0.0015430269,0.0008267845,0.0007914829],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005653375,0.0002068543,0.0049751387,0.00025967322,0.00010330515,0.00044783935,0.0003142112,0.37015182,0.041382324,0.042954586,0.0044835606,0.5341553],"study_design_scores_gemma":[0.00006293032,0.00020855058,0.0008794965,0.00003234492,0.000045490055,0.000586614,0.000057842637,0.9487537,0.018684963,0.026489351,0.004156231,0.000042401836],"about_ca_topic_score_codex":0.0017595842,"about_ca_topic_score_gemma":0.0018420289,"teacher_disagreement_score":0.0024937452,"about_ca_system_score_codex":0.00074816693,"about_ca_system_score_gemma":0.0010455183,"threshold_uncertainty_score":0.008342385},"labels":[],"label_agreement":null},{"id":"W4254309703","doi":"10.1007/s10270-002-8208-5","title":"A UML-Based Approach to System Testing","year":2002,"lang":"en","type":"article","venue":"Software & Systems Modeling","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Unified Modeling Language; Sequence diagram; Class diagram; Object Constraint Language; Software engineering; Applications of UML; Testability; Model-based testing; Context (archaeology); Test case; Programming language; UML tool; Test Management Approach; System testing; Activity diagram; Software development; Reliability engineering; Software; Software construction","score_opus":0.09020794658056883,"score_gpt":0.23924228661756475,"score_spread":0.14903434003699592,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254309703","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008865838,0.00014151701,0.9941871,0.0003244034,0.00004083728,0.00006556288,0.00002652638,0.0009066582,0.0034207378],"genre_scores_gemma":[0.051211227,0.0003507326,0.9442317,0.00029585356,0.00007275666,0.0001947028,0.00015004887,0.00033551131,0.0031574618],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9917076,0.0035017417,0.0005244516,0.0004615456,0.00353784,0.00026677919],"domain_scores_gemma":[0.99087137,0.0048828493,0.00037015523,0.0020897086,0.0015608786,0.0002250476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005754771,0.0011409867,0.00083144783,0.0032379662,0.0010371004,0.0034313796,0.0032840506,0.0019970092,0.0053855916],"category_scores_gemma":[0.01883765,0.001087818,0.0016475619,0.0014893862,0.0022814472,0.0053011933,0.0025249429,0.003775695,0.0016036974],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007865132,0.00032912387,0.0010540042,0.00033714398,0.00010160663,0.000330659,0.0007536746,0.042095844,0.009260534,0.6227638,0.0074450416,0.31544995],"study_design_scores_gemma":[0.00008856354,0.00014779656,0.00067070586,0.00044707267,0.00016927061,0.000826291,0.00015577664,0.4362407,0.013475786,0.47919646,0.06850389,0.000077723336],"about_ca_topic_score_codex":0.003171894,"about_ca_topic_score_gemma":0.0046499404,"teacher_disagreement_score":0.005754771,"about_ca_system_score_codex":0.0013859853,"about_ca_system_score_gemma":0.0019593458,"threshold_uncertainty_score":0.03043449},"labels":[],"label_agreement":null},{"id":"W4254402004","doi":"10.22215/etd/2010-09070","title":"Systematic review of state based model based testing tools","year":2010,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Library and Archives Canada","funders":"","keywords":"State (computer science); Computer science; Programming language","score_opus":0.03741510849190186,"score_gpt":0.2970245142930462,"score_spread":0.25960940580114433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254402004","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049907113,0.9785211,0.009143518,0.0017720424,0.0005552942,0.0018917293,0.0019639588,0.00014062603,0.0010210414],"genre_scores_gemma":[0.07278174,0.88467616,0.03121977,0.0024112759,0.00025789245,0.0049886736,0.002758283,0.000114247036,0.0007919872],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.96217,0.015290348,0.014227009,0.0018992841,0.0059641725,0.00044923712],"domain_scores_gemma":[0.751824,0.19386551,0.030419264,0.0069734226,0.015638884,0.0012788723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.031314116,0.0015968006,0.0050245454,0.02331299,0.00078933855,0.0033656706,0.0032183882,0.002113533,0.0040492215],"category_scores_gemma":[0.14488354,0.0011930917,0.0067308815,0.01435755,0.0013156221,0.0041895146,0.003001578,0.0011995747,0.00046406014],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085619476,0.00011389456,0.0018850663,0.63197273,0.008270788,0.00025854987,0.0006991135,0.00078228174,0.0011995999,0.0013328729,0.006018607,0.3466103],"study_design_scores_gemma":[0.0011404051,0.0015745268,0.008384316,0.8486884,0.05692623,0.00091188133,0.0006751749,0.0006716095,0.0018827302,0.0030112509,0.076000385,0.00013308415],"about_ca_topic_score_codex":0.004733028,"about_ca_topic_score_gemma":0.019495444,"teacher_disagreement_score":0.031314116,"about_ca_system_score_codex":0.0033689027,"about_ca_system_score_gemma":0.019374978,"threshold_uncertainty_score":0.16560686},"labels":[],"label_agreement":null},{"id":"W4254576214","doi":"10.1109/icse.2000.870402","title":"Broad-spectrum studies of log file analysis","year":2002,"lang":"en","type":"article","venue":"Proceedings of the 2000 International Conference on Software Engineering. ICSE 2000 the New Millennium","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; File system; Completeness (order theory); Unix file types; Computer file; Operating system; Stub file; Mathematics","score_opus":0.03605241013301291,"score_gpt":0.25861406051296104,"score_spread":0.22256165037994813,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254576214","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14354499,0.06174034,0.7164152,0.0042390004,0.00036783176,0.00036838782,0.00018846332,0.0009461546,0.0721896],"genre_scores_gemma":[0.76197857,0.037579075,0.19338746,0.0010965418,0.0007451138,0.00032020183,0.0002989312,0.00043408378,0.004160029],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9866159,0.006188983,0.00068769074,0.0012083398,0.0048441254,0.00045497058],"domain_scores_gemma":[0.79924816,0.17170049,0.0036266574,0.008946528,0.015411846,0.0010663837],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0103397155,0.00090169284,0.0010406981,0.011111885,0.0017277893,0.005028841,0.0028554774,0.0023666068,0.0038008653],"category_scores_gemma":[0.09686238,0.0009349657,0.00089639897,0.009366061,0.0039260443,0.013450179,0.003221331,0.0032849216,0.0006869605],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052573334,0.00060554605,0.01907841,0.0023062772,0.00012158501,0.00048048518,0.005141132,0.011515198,0.0072619733,0.29445493,0.0027674045,0.65574133],"study_design_scores_gemma":[0.0001875555,0.0013377247,0.041087378,0.008995301,0.00034318626,0.008845582,0.01131575,0.19380224,0.03911976,0.50390667,0.19055294,0.0005059857],"about_ca_topic_score_codex":0.0017500934,"about_ca_topic_score_gemma":0.0011381541,"teacher_disagreement_score":0.011111885,"about_ca_system_score_codex":0.0020190312,"about_ca_system_score_gemma":0.0015130291,"threshold_uncertainty_score":0.054682314},"labels":[],"label_agreement":null},{"id":"W4255107725","doi":"10.1007/978-1-4939-7131-2_101286","title":"Subgraph Identification","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Identification (biology); Computer science; Biology; Botany","score_opus":0.02745455742923722,"score_gpt":0.25097130965708847,"score_spread":0.22351675222785125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4255107725","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007843584,0.0016158727,0.7914125,0.0008554111,0.0007037846,0.00032050244,0.0031998006,0.0103559205,0.18369265],"genre_scores_gemma":[0.10157599,0.0029937641,0.6625464,0.00064618915,0.00040747176,0.00031174056,0.019283919,0.0044226786,0.20781182],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99935347,0.00007560454,0.00002424772,0.00025637835,0.00022525714,0.0000649663],"domain_scores_gemma":[0.9991836,0.00014861191,0.00003819437,0.00035746285,0.00022346298,0.00004865309],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00037816894,0.0012195684,0.00088007154,0.0028006101,0.001016906,0.0017454738,0.0014585318,0.0008501829,0.04818911],"category_scores_gemma":[0.0017767308,0.0005068852,0.0010585443,0.002511252,0.0006582487,0.0029949055,0.0018936953,0.0014007337,0.02850587],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000074412164,0.000074408825,0.0006465531,0.00043600568,0.00003916006,0.00015629061,0.00015349587,0.0051725702,0.009863498,0.111692496,0.13432978,0.7373613],"study_design_scores_gemma":[0.000023425511,0.000065672284,0.0015436903,0.00027120218,0.00009233391,0.0014109106,0.00030893585,0.055553753,0.035259392,0.3352395,0.57016504,0.00006621052],"about_ca_topic_score_codex":0.0015205902,"about_ca_topic_score_gemma":0.0029386783,"teacher_disagreement_score":0.04818911,"about_ca_system_score_codex":0.0007158496,"about_ca_system_score_gemma":0.0010486955,"threshold_uncertainty_score":0.16120863},"labels":[],"label_agreement":null},{"id":"W4255224654","doi":"10.22215/etd/2007-07874","title":"An empirical study of the regression testing of an industrial software product","year":2007,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage","funders":"","keywords":"Product (mathematics); Software; Computer science; Statistics; Engineering; Mathematics; Programming language","score_opus":0.09995236811702643,"score_gpt":0.3939676865431514,"score_spread":0.29401531842612494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4255224654","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99729496,0.00013942784,0.0008141628,0.000069013295,0.0000052417117,0.000028295946,0.00010727713,0.000019669764,0.0015218398],"genre_scores_gemma":[0.99824774,0.00007484133,0.0007682427,0.000028147038,0.000007257465,0.00002492862,0.00029979306,0.000008185017,0.00054081087],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9942106,0.0035244997,0.0003769714,0.00046907246,0.0011521064,0.0002666627],"domain_scores_gemma":[0.6564631,0.28094748,0.038243394,0.00801864,0.012707388,0.0036199521],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007453384,0.00041879647,0.00018560019,0.0016601597,0.00045097328,0.0008217022,0.0011254745,0.00075722113,0.0025551552],"category_scores_gemma":[0.1120078,0.00021212226,0.0002638459,0.0014720263,0.0009959759,0.0014540836,0.0005205351,0.0012946788,0.0008329616],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007198223,0.0035296148,0.95577806,0.00013515029,0.000104513165,0.00047348673,0.0018531905,0.0013952197,0.0012248736,0.0010283324,0.00085306575,0.032904726],"study_design_scores_gemma":[0.00006017011,0.0039936183,0.9713851,0.000076286335,0.000084769505,0.0009421579,0.003109207,0.015277797,0.0018988944,0.0004806405,0.0026572528,0.000034088804],"about_ca_topic_score_codex":0.004099647,"about_ca_topic_score_gemma":0.0033300682,"teacher_disagreement_score":0.007453384,"about_ca_system_score_codex":0.0006280198,"about_ca_system_score_gemma":0.00053278834,"threshold_uncertainty_score":0.039417744},"labels":[],"label_agreement":null},{"id":"W4280619266","doi":"10.4230/lipics.itp.2022.18","title":"Automatic Test-Case Reduction in Proof Assistants: A Case Study in Coq","year":2022,"lang":"en","type":"article","venue":"DROPS (Schloss Dagstuhl – Leibniz Center for Informatics)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Prevention of Organ Failure","funders":"","keywords":"Computer science; Reduction (mathematics); Test (biology); Proof assistant; Proof of concept; Programming language; Operating system; Mathematics; Mathematical proof","score_opus":0.02613850071469267,"score_gpt":0.29371747690243355,"score_spread":0.2675789761877409,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4280619266","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7600886,0.0007399402,0.22384034,0.0007820198,0.00004994213,0.0005429865,0.00031512303,0.009326984,0.004314043],"genre_scores_gemma":[0.7809086,0.0002265254,0.21392475,0.00021481753,0.000020959737,0.00020663625,0.00047021182,0.0020340232,0.0019935253],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9827209,0.010059453,0.00088253606,0.0015209692,0.0039800205,0.00083610794],"domain_scores_gemma":[0.8032766,0.16475502,0.0045119002,0.017596744,0.008369414,0.0014902845],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013411,0.0010472698,0.00070713134,0.0013448981,0.0011041978,0.0014673,0.0030660566,0.001995211,0.0018146044],"category_scores_gemma":[0.08403687,0.0008621739,0.0006981056,0.0014138462,0.0022347737,0.0025272546,0.0019770532,0.0021623033,0.00065129803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031486766,0.0054241894,0.04792395,0.0024992872,0.00028081491,0.010156485,0.023536662,0.15122975,0.07847746,0.027817369,0.015248625,0.6342567],"study_design_scores_gemma":[0.001359443,0.0060502803,0.029312355,0.00056060345,0.00032726882,0.011517374,0.004581789,0.6708084,0.17884082,0.017186088,0.07912508,0.00033057661],"about_ca_topic_score_codex":0.005994147,"about_ca_topic_score_gemma":0.0050094123,"teacher_disagreement_score":0.013411,"about_ca_system_score_codex":0.001289634,"about_ca_system_score_gemma":0.001534885,"threshold_uncertainty_score":0.070925},"labels":[],"label_agreement":null},{"id":"W4280650798","doi":"10.18280/ria.360217","title":"Combinatorial Test Case Generation Using Q-Value Based Particle Swarm Optimization","year":2022,"lang":"en","type":"article","venue":"Revue d intelligence artificielle","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Particle swarm optimization; Pairwise comparison; Mathematical optimization; Computer science; Test case; Heuristic; Metaheuristic; Set (abstract data type); Value (mathematics); Fitness function; Function (biology); Swarm behaviour; Algorithm; Mathematics; Artificial intelligence; Machine learning; Genetic algorithm","score_opus":0.07369328942555042,"score_gpt":0.2950321892962207,"score_spread":0.2213388998706703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4280650798","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026298534,0.00022228096,0.9679126,0.00017548942,0.000049049544,0.00049080426,0.000062782725,0.0006939708,0.004094604],"genre_scores_gemma":[0.48586047,0.00023177087,0.51093626,0.00014125048,0.000024378644,0.0006951976,0.0002568025,0.000099268276,0.0017545837],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985273,0.0005949683,0.000090957335,0.00017845033,0.0004856138,0.0001226194],"domain_scores_gemma":[0.99604744,0.0028175178,0.00027660961,0.00020281301,0.0005730648,0.00008248376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019234903,0.0012588516,0.0010523434,0.0016567806,0.0005095012,0.00109219,0.0015035965,0.0011817237,0.002249451],"category_scores_gemma":[0.006625563,0.0005704187,0.0010599651,0.0011899916,0.00077637506,0.000819052,0.00090443256,0.0009810157,0.00028973926],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007497473,0.00011318677,0.0009450649,0.000104961044,0.0000489629,0.00014753106,0.000059812362,0.9182437,0.002461475,0.0072499868,0.0009138524,0.069636405],"study_design_scores_gemma":[0.000026400303,0.00006057786,0.00010041504,0.00000705341,0.000008956444,0.000025291169,0.000007915991,0.99662423,0.0011793689,0.0014841754,0.00047005006,0.000005501233],"about_ca_topic_score_codex":0.0047573624,"about_ca_topic_score_gemma":0.0031262494,"teacher_disagreement_score":0.0047573624,"about_ca_system_score_codex":0.0011070137,"about_ca_system_score_gemma":0.0012822754,"threshold_uncertainty_score":0.010172486},"labels":[],"label_agreement":null},{"id":"W4281765649","doi":"10.1016/j.cca.2022.04.224","title":"M126 Effective utilization management strategies to limit inappropriate referred-out test requests","year":2022,"lang":"en","type":"article","venue":"Clinica Chimica Acta","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Health Canada; Nova Scotia Health Authority; Dalhousie University","funders":"","keywords":"Limit (mathematics); Test (biology); Utilization management; Operations management; Business; Intensive care medicine; Medicine; Risk analysis (engineering); Engineering; Mathematics; Political science; Health care; Biology; Law","score_opus":0.06263412735368994,"score_gpt":0.3412593811587694,"score_spread":0.27862525380507946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281765649","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30383188,0.0054096314,0.49370953,0.035033744,0.0014494089,0.002673347,0.0017710478,0.035367142,0.12075419],"genre_scores_gemma":[0.8747409,0.0006653575,0.10786259,0.002845897,0.00042998226,0.00058864604,0.00068881287,0.00047206166,0.0117058605],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9965306,0.0012568183,0.000386868,0.00033986673,0.0011677858,0.00031812364],"domain_scores_gemma":[0.98816967,0.0035095196,0.0026370077,0.0010659209,0.003292452,0.0013252995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00257313,0.0008066188,0.00041808368,0.0026860647,0.00107302,0.0023704334,0.0018359353,0.0013398412,0.010360068],"category_scores_gemma":[0.018493675,0.0003055254,0.0004109124,0.0010299541,0.00033489006,0.001236266,0.00128825,0.0011271794,0.0026063863],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057234097,0.001761437,0.05198894,0.00024953595,0.00007240852,0.00090212194,0.0005250909,0.0072157606,0.012754826,0.004824708,0.060245283,0.85888755],"study_design_scores_gemma":[0.0008662766,0.004726488,0.2589379,0.0026168549,0.00064636377,0.011372466,0.0051427376,0.389753,0.080205366,0.04266819,0.20255516,0.0005091825],"about_ca_topic_score_codex":0.0027037363,"about_ca_topic_score_gemma":0.0034082376,"teacher_disagreement_score":0.010360068,"about_ca_system_score_codex":0.0013010832,"about_ca_system_score_gemma":0.0035008476,"threshold_uncertainty_score":0.034657836},"labels":[],"label_agreement":null},{"id":"W4282828008","doi":"10.1109/icse-companion55297.2022.9793739","title":"DScribe: Co-generating Unit Tests and Documentation","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/ACM 44th International Conference on Software Engineering: Companion Proceedings (ICSE-Companion)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Nature","keywords":"Documentation; Computer science; Unit testing; Unit (ring theory); Operating system; Psychology; Software","score_opus":0.04851208535317507,"score_gpt":0.2975137824159923,"score_spread":0.24900169706281722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4282828008","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007847993,0.00020392954,0.9135712,0.00027606403,0.00018094556,0.00050528813,0.0007756849,0.07019568,0.0064432016],"genre_scores_gemma":[0.09454739,0.00021401838,0.86995345,0.00040441714,0.000087903856,0.000710387,0.004693608,0.01996125,0.009427453],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9909464,0.0028949585,0.0006129869,0.0013842913,0.0037708853,0.00039050533],"domain_scores_gemma":[0.952353,0.01897915,0.0022345542,0.018921234,0.006558235,0.0009537744],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0076903724,0.0019195525,0.0010489229,0.0044179275,0.00062353775,0.0030162025,0.0039627217,0.0020264632,0.0131148575],"category_scores_gemma":[0.057023212,0.0015321136,0.0018115289,0.001960398,0.0016182784,0.003517363,0.006513863,0.0025409176,0.0071138297],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063444086,0.00063669565,0.008505187,0.0009813844,0.00024457235,0.0014736217,0.001331023,0.028704314,0.024015665,0.03243906,0.06284122,0.83819276],"study_design_scores_gemma":[0.00057501695,0.0006200341,0.003407805,0.0004955994,0.00018864528,0.0029783729,0.00048395497,0.56361294,0.17119327,0.05618596,0.19995059,0.00030784853],"about_ca_topic_score_codex":0.0014188794,"about_ca_topic_score_gemma":0.002400141,"teacher_disagreement_score":0.0131148575,"about_ca_system_score_codex":0.000951955,"about_ca_system_score_gemma":0.00224749,"threshold_uncertainty_score":0.043873608},"labels":[],"label_agreement":null},{"id":"W4282830605","doi":"10.1109/icse-companion55297.2022.9793757","title":"DiffWatch: Watch Out for the Evolving Differential Testing in Deep Learning Libraries","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/ACM 44th International Conference on Software Engineering: Companion Proceedings (ICSE-Companion)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Python (programming language); Computer science; Differential (mechanical device); Deep learning; Artificial intelligence; Operating system; Engineering","score_opus":0.05555531626052307,"score_gpt":0.26780240175017994,"score_spread":0.21224708548965687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4282830605","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06965683,0.0011519701,0.23329817,0.0031748184,0.00070474483,0.0005209165,0.004866465,0.67616653,0.010459545],"genre_scores_gemma":[0.6606765,0.0006951561,0.24002524,0.0041272533,0.00024788515,0.0010006889,0.009988563,0.06688268,0.016356101],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99454474,0.001502125,0.0004880501,0.0010743673,0.0018225946,0.0005680268],"domain_scores_gemma":[0.9701855,0.016152171,0.0022165414,0.007583562,0.0026371453,0.0012250737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007657402,0.0020652893,0.0006658867,0.0033657376,0.0007296712,0.0021246832,0.003738379,0.0015555514,0.010163708],"category_scores_gemma":[0.042583898,0.0016037348,0.00080981496,0.0010484663,0.0017209117,0.0071389964,0.005338734,0.0033921117,0.0038296336],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020091606,0.00053863565,0.06485463,0.0011135045,0.00019985267,0.0021953825,0.0032791218,0.009508835,0.022335034,0.010690638,0.25934625,0.62392896],"study_design_scores_gemma":[0.0008781738,0.0015895368,0.056439575,0.0013466334,0.00022960306,0.00325752,0.0012613473,0.38753396,0.15782589,0.055770766,0.33296034,0.0009066405],"about_ca_topic_score_codex":0.0034348315,"about_ca_topic_score_gemma":0.005208088,"teacher_disagreement_score":0.010163708,"about_ca_system_score_codex":0.0013243625,"about_ca_system_score_gemma":0.0017623144,"threshold_uncertainty_score":0.040496707},"labels":[],"label_agreement":null},{"id":"W4282835086","doi":"10.1109/icse-companion55297.2022.9793738","title":"MASS: A tool for Mutation Analysis of Space CPS","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/ACM 44th International Conference on Software Engineering: Companion Proceedings (ICSE-Companion)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"European Research Council; Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Scalability; Computer science; Context (archaeology); Software; Set (abstract data type); Mutation; Operating system; Programming language","score_opus":0.036619287075395565,"score_gpt":0.2779605335269983,"score_spread":0.2413412464516027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4282835086","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008376435,0.000421115,0.7993476,0.00023330198,0.00010574193,0.00022150227,0.002763903,0.18391012,0.0046203267],"genre_scores_gemma":[0.18839285,0.00074347074,0.76322854,0.00050622586,0.00012032647,0.001274064,0.0091358675,0.028620986,0.007977733],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986457,0.00027069607,0.00012701335,0.00023022258,0.0006074501,0.000118799806],"domain_scores_gemma":[0.9962858,0.0025604526,0.00038528896,0.00035455174,0.00034144448,0.000072449315],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017712185,0.0024292264,0.00096231943,0.0034046553,0.00068349455,0.0017335906,0.0021772254,0.0014921033,0.017805433],"category_scores_gemma":[0.008871464,0.0010666476,0.0019751724,0.0012618348,0.0010077971,0.002590987,0.0022563287,0.0016791646,0.0039809956],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00084183976,0.00043690903,0.015015566,0.0027054525,0.0006125169,0.0027837257,0.0011368837,0.19797035,0.038515862,0.06750591,0.16453354,0.5079414],"study_design_scores_gemma":[0.00026663428,0.00024455047,0.0024114337,0.00038884216,0.00013215665,0.0014601266,0.00016797966,0.8257702,0.031210553,0.055298034,0.08250872,0.00014080218],"about_ca_topic_score_codex":0.0022349323,"about_ca_topic_score_gemma":0.0026268512,"teacher_disagreement_score":0.017805433,"about_ca_system_score_codex":0.00074635004,"about_ca_system_score_gemma":0.0013029564,"threshold_uncertainty_score":0.059565067},"labels":[],"label_agreement":null},{"id":"W4283080616","doi":"10.1109/icse-seip55303.2022.9793941","title":"The Impact of Flaky Tests on Historical Test Prioritization on Chrome","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Prioritization; Blocking (statistics); Pipeline (software); Computer science; Test (biology); Reliability engineering; Replication (statistics); Work (physics); Regression testing; Software; Engineering; Software development; Operating system; Computer network","score_opus":0.019321535303824963,"score_gpt":0.28688465220610365,"score_spread":0.2675631169022787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283080616","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.81642544,0.0033236342,0.1068947,0.0019410646,0.00044390728,0.00037780474,0.0016204364,0.056607287,0.01236577],"genre_scores_gemma":[0.9000005,0.00028058584,0.0936924,0.00043914016,0.000045851408,0.000078107594,0.0016872432,0.0020306294,0.0017455325],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98345643,0.0045012753,0.0010819222,0.0035598695,0.006347861,0.0010527761],"domain_scores_gemma":[0.87099695,0.081582725,0.0056937914,0.026251545,0.013280495,0.0021945534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012666906,0.0014187085,0.00081594195,0.0025817845,0.0010274285,0.0028606348,0.0028612835,0.0012170011,0.0018027537],"category_scores_gemma":[0.110576645,0.00093346136,0.0007150696,0.0021326807,0.001830224,0.004787216,0.0016148431,0.0025756902,0.00071447797],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024641776,0.0011121376,0.15850897,0.0007542929,0.00040185856,0.00040400735,0.0009556048,0.16571844,0.032130096,0.0076071187,0.028194116,0.6017492],"study_design_scores_gemma":[0.00035831323,0.0017972025,0.06837135,0.00018940087,0.00023974502,0.0007322407,0.0004920764,0.84433454,0.06308258,0.0058188164,0.014342313,0.00024149223],"about_ca_topic_score_codex":0.022919256,"about_ca_topic_score_gemma":0.023151174,"teacher_disagreement_score":0.022919256,"about_ca_system_score_codex":0.0027530456,"about_ca_system_score_gemma":0.0026057246,"threshold_uncertainty_score":0.06698984},"labels":[],"label_agreement":null},{"id":"W4283323653","doi":"10.1145/3544790","title":"Toward More Efficient Statistical Debugging with Abstraction Refinement","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Office of Naval Research; Natural Science Foundation of Jiangsu Province; National Natural Science Foundation of China; National Science Foundation","keywords":"Debugging; Computer science; Algorithmic program debugging; Abstraction; Programming language; Pruning; Discriminative model; Software engineering; Machine learning","score_opus":0.07935506400377887,"score_gpt":0.3095148429020107,"score_spread":0.23015977889823183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283323653","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007685265,0.0001368255,0.9894302,0.00015760402,0.00001411993,0.00006014936,0.000036952133,0.0021636565,0.0003152491],"genre_scores_gemma":[0.20179886,0.00024081589,0.79577863,0.00028281342,0.00004325046,0.00019988991,0.00032588057,0.0007381718,0.00059171],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98449576,0.0062291147,0.0009504119,0.0015206581,0.0059493096,0.000854819],"domain_scores_gemma":[0.9642211,0.016460415,0.0032757698,0.01123442,0.00444098,0.00036729133],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010319101,0.0015169284,0.0014262837,0.0027609447,0.0009093215,0.0017841392,0.0031483236,0.0010307786,0.0011319132],"category_scores_gemma":[0.04409249,0.0010501927,0.0017673941,0.0018867197,0.0017630114,0.0040756967,0.0037183862,0.0031970886,0.0007225902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004855957,0.00040742752,0.019159801,0.00060663524,0.00031698158,0.0006652528,0.0012804056,0.26398683,0.07003285,0.09568154,0.006852462,0.54052424],"study_design_scores_gemma":[0.00008183279,0.0002103275,0.0016877621,0.00009864586,0.000115066665,0.00036221617,0.00013220758,0.8957353,0.02301731,0.0720583,0.006444765,0.00005614976],"about_ca_topic_score_codex":0.0026624925,"about_ca_topic_score_gemma":0.004811287,"teacher_disagreement_score":0.010319101,"about_ca_system_score_codex":0.0009712569,"about_ca_system_score_gemma":0.004262997,"threshold_uncertainty_score":0.054573238},"labels":[],"label_agreement":null},{"id":"W4283689178","doi":"10.1016/j.cose.2022.102813","title":"Fuzzing vulnerability discovery techniques: Survey, challenges and future directions","year":2022,"lang":"en","type":"article","venue":"Computers & Security","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Fuzz testing; Computer science; Vulnerability (computing); Computer security; Data science; Programming language","score_opus":0.027466111279944364,"score_gpt":0.26526658250549223,"score_spread":0.23780047122554787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283689178","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016421588,0.64295626,0.31869993,0.010960425,0.00047005972,0.00034739554,0.00025579528,0.0014713161,0.008417206],"genre_scores_gemma":[0.11149056,0.51848334,0.36196476,0.0019731263,0.0014332503,0.00025025354,0.00087173143,0.00032991564,0.0032030162],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99082583,0.0019304352,0.001022745,0.0017798424,0.004009556,0.00043155448],"domain_scores_gemma":[0.951719,0.03307408,0.0019620003,0.00461781,0.007959305,0.00066778914],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015835334,0.0022124387,0.0031860191,0.009010043,0.0011548931,0.005409896,0.006334835,0.00309805,0.002980962],"category_scores_gemma":[0.025253713,0.0015511474,0.002055162,0.0060055624,0.002677129,0.015928322,0.002979062,0.0047854763,0.0012497182],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000104497216,0.00035320222,0.0051861275,0.0038417939,0.00016238463,0.000106590975,0.00037611378,0.005805806,0.0031688255,0.027332503,0.0052039404,0.94835824],"study_design_scores_gemma":[0.00022999557,0.0020524047,0.010331496,0.020892542,0.0011671465,0.0072276155,0.0031688863,0.26463538,0.028112922,0.3316904,0.32995656,0.00053467415],"about_ca_topic_score_codex":0.0018736736,"about_ca_topic_score_gemma":0.0023251018,"teacher_disagreement_score":0.015835334,"about_ca_system_score_codex":0.0013628133,"about_ca_system_score_gemma":0.003002738,"threshold_uncertainty_score":0.083746254},"labels":[],"label_agreement":null},{"id":"W4284682233","doi":"10.1145/3510003.3510188","title":"Efficient online testing for DNN-enabled systems using surrogate-assisted and many-objective optimization","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 44th International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; National Research Foundation of Korea; Fonds National de la Recherche Luxembourg; National Research Foundation; European Commission","keywords":"Computer science; Reliability (semiconductor); Fidelity; High fidelity; Reliability engineering; Software; Test strategy; Artificial neural network; Embedded system; Real-time computing; Distributed computing; Machine learning; Engineering; Operating system","score_opus":0.04979620041711109,"score_gpt":0.266732807629432,"score_spread":0.21693660721232094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4284682233","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10929246,0.0011185618,0.8798147,0.0008583917,0.000117766656,0.00017255131,0.00024090036,0.0026924212,0.005692321],"genre_scores_gemma":[0.87648654,0.00015071018,0.12028751,0.00026808053,0.000025060746,0.0001995299,0.00042879186,0.00026423737,0.0018894714],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99878603,0.0005264807,0.000063691296,0.00017253358,0.00025335772,0.00019788374],"domain_scores_gemma":[0.9929404,0.005540471,0.0004069603,0.00027644375,0.00058950426,0.0002462435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002368916,0.0018167885,0.0015662586,0.00070412847,0.00040500137,0.0010248222,0.0018149767,0.0016607447,0.0034910997],"category_scores_gemma":[0.009252509,0.0008935453,0.0007874873,0.00038566228,0.0012534925,0.0014002003,0.0015804176,0.0022037935,0.00041001593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000093351286,0.000045414286,0.0005710026,0.00007457626,0.000024814472,0.000053769236,0.000014940208,0.983582,0.00057414756,0.0014686303,0.00039309182,0.013104329],"study_design_scores_gemma":[0.000005755069,0.000015573194,0.000034668574,0.0000045719526,0.0000020636433,0.0000045262827,0.000002764005,0.9986965,0.00017817308,0.0010046117,0.000049488914,0.0000014021743],"about_ca_topic_score_codex":0.0072391415,"about_ca_topic_score_gemma":0.009303014,"teacher_disagreement_score":0.0072391415,"about_ca_system_score_codex":0.0015508913,"about_ca_system_score_gemma":0.00208294,"threshold_uncertainty_score":0.014393985},"labels":[],"label_agreement":null},{"id":"W4284693828","doi":"10.1145/3510003.3510106","title":"Nessie","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 44th International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Callback; Computer science; Asynchronous communication; Unit testing; Test case; Generator (circuit theory); JavaScript; Test (biology); Operating system; Programming language; Software; Machine learning; Computer network","score_opus":0.02496960565977006,"score_gpt":0.24090047315395033,"score_spread":0.21593086749418028,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4284693828","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006846061,0.00035096772,0.9224019,0.00022568487,0.0001663431,0.00040137253,0.0015616381,0.048626043,0.019419959],"genre_scores_gemma":[0.08142891,0.00032544564,0.8785304,0.00041208035,0.000046902776,0.0006197538,0.0069701294,0.0074432436,0.024223166],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9973827,0.00062712363,0.00023139734,0.0005701202,0.0009875647,0.00020111245],"domain_scores_gemma":[0.9950222,0.0019246807,0.00036034748,0.0012873526,0.0012543979,0.00015109591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021166892,0.0014657374,0.00081524404,0.0019434107,0.00064628583,0.0017063366,0.0031761047,0.0012492231,0.028581657],"category_scores_gemma":[0.010058254,0.00079900096,0.0014057852,0.0009766886,0.0010226449,0.0024696775,0.0023476735,0.0017002036,0.014735238],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008457471,0.00035550585,0.0037481552,0.0013897289,0.00012501978,0.0004925689,0.00041825153,0.052025672,0.021915236,0.10462277,0.0651587,0.7489026],"study_design_scores_gemma":[0.00026890705,0.000530723,0.0011923198,0.00032844706,0.000100435806,0.001445174,0.0001589082,0.5655947,0.056740783,0.093885325,0.27963868,0.00011566629],"about_ca_topic_score_codex":0.0018355498,"about_ca_topic_score_gemma":0.0038669715,"teacher_disagreement_score":0.028581657,"about_ca_system_score_codex":0.0011152098,"about_ca_system_score_gemma":0.0019178286,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4285490476","doi":"10.1145/3533767.3543293","title":"ATUA: an update-driven app testing tool","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"European Commission","keywords":"Computer science; Oracle; Android (operating system); Random testing; Code coverage; Source code; Regression testing; Model-based testing; Code (set theory); Test case; Software engineering; Programming language; Machine learning; Software; Operating system; Software development; Regression analysis","score_opus":0.03478549762453586,"score_gpt":0.26095904739117276,"score_spread":0.2261735497666369,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285490476","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040759094,0.0010187675,0.53633446,0.00037951348,0.0001876853,0.0004831979,0.0028589582,0.40995398,0.008024239],"genre_scores_gemma":[0.5516866,0.0009010433,0.39580697,0.00079642166,0.0001305088,0.0011537814,0.009275357,0.029915486,0.010333843],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.997976,0.00045013087,0.0001706755,0.00030496294,0.00095800514,0.0001401775],"domain_scores_gemma":[0.9932307,0.004199559,0.00043372886,0.0010636861,0.0008971161,0.00017516996],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014052925,0.0016254311,0.00073279516,0.0025107658,0.00041926326,0.0012440141,0.0028403548,0.0012367012,0.006736523],"category_scores_gemma":[0.0126331365,0.0008531132,0.0011718146,0.00070919207,0.0007230586,0.002307021,0.0021952237,0.0012773777,0.002925576],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012772325,0.0005749608,0.01979517,0.0020001621,0.00036370655,0.0029888358,0.0016812633,0.05485225,0.07038324,0.01527743,0.1104272,0.7203786],"study_design_scores_gemma":[0.00033207127,0.00091607985,0.008171038,0.00056419655,0.00033341552,0.0041456437,0.0003840459,0.733303,0.10190685,0.021487357,0.12810346,0.0003528195],"about_ca_topic_score_codex":0.00287646,"about_ca_topic_score_gemma":0.0025785374,"teacher_disagreement_score":0.006736523,"about_ca_system_score_codex":0.00043603938,"about_ca_system_score_gemma":0.001155154,"threshold_uncertainty_score":0.02253592},"labels":[],"label_agreement":null},{"id":"W4288406204","doi":"10.48550/arxiv.1903.11242","title":"An Empirical Study on Practicality of Specification Mining Algorithms on\\n a Real-world Application","year":2019,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Debugging; Computer science; Program comprehension; Context (archaeology); Implementation; Inference; Programming language; Abstraction; Software engineering; Set (abstract data type); Root cause; Software; Algorithm; Artificial intelligence; Software system; Reliability engineering","score_opus":0.2026427942074081,"score_gpt":0.31376238393988776,"score_spread":0.11111958973247965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288406204","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97129345,0.0015796989,0.017800266,0.0014579393,0.00007743448,0.00025714055,0.0015563355,0.0008663319,0.0051113074],"genre_scores_gemma":[0.97732496,0.0003620272,0.018451966,0.00012522463,0.000037796337,0.0001319939,0.0027502025,0.00017109192,0.0006447505],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9740233,0.014932273,0.0025294349,0.0036047385,0.004198044,0.00071221945],"domain_scores_gemma":[0.4254025,0.5207703,0.010075633,0.030527726,0.01128624,0.0019375245],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.028348742,0.0008949707,0.0006463881,0.0022258635,0.0008902073,0.0024031308,0.0020943775,0.001808004,0.0027677484],"category_scores_gemma":[0.28851008,0.00048774458,0.00094119355,0.0029848432,0.002097559,0.005335395,0.0016742775,0.0027452752,0.0010301105],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0056820544,0.0060783564,0.42645583,0.0024456875,0.0009060954,0.00053600775,0.003547087,0.11353927,0.006807907,0.01115452,0.017830925,0.4050162],"study_design_scores_gemma":[0.0007815501,0.0038405391,0.15375139,0.00047407276,0.0003759144,0.0014373577,0.0038624045,0.788905,0.01280497,0.014442823,0.019169865,0.0001540323],"about_ca_topic_score_codex":0.0024673573,"about_ca_topic_score_gemma":0.002893594,"teacher_disagreement_score":0.028348742,"about_ca_system_score_codex":0.001686851,"about_ca_system_score_gemma":0.0013599752,"threshold_uncertainty_score":0.14992428},"labels":[],"label_agreement":null},{"id":"W4288723601","doi":"10.1007/s10664-022-10158-x","title":"GBGallery : A benchmark and framework for game testing","year":2022,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Benchmark (surveying); Software engineering; Video game development; Test strategy; Software development; Game testing; Software; Database; Game Developer; Game design; Artificial intelligence; Programming language; Game design document","score_opus":0.04408833245119895,"score_gpt":0.28456053815274496,"score_spread":0.240472205701546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288723601","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013168616,0.00062440627,0.83877474,0.0009731802,0.0002669347,0.0005381795,0.0039051012,0.12838228,0.013366644],"genre_scores_gemma":[0.1539991,0.0005101744,0.8125084,0.00062160194,0.000114262846,0.0011647376,0.00998501,0.017035436,0.0040612016],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9882515,0.005074975,0.001466504,0.0008703926,0.0034686401,0.0008679527],"domain_scores_gemma":[0.96199566,0.02177425,0.0018917749,0.008415544,0.004518898,0.0014038333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010397868,0.0025978116,0.0013873264,0.006183542,0.0009811064,0.004377573,0.007758392,0.002997756,0.009800072],"category_scores_gemma":[0.071076915,0.0012909091,0.0015327097,0.0035188543,0.0019065775,0.006404637,0.005151144,0.0043045147,0.0042933375],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012402228,0.0014061301,0.015018423,0.0018185867,0.00028010385,0.0005642026,0.0006934455,0.11018866,0.007305918,0.20291373,0.17974636,0.47882417],"study_design_scores_gemma":[0.000447403,0.0004896988,0.0034398285,0.00069471967,0.00010081218,0.000743031,0.00022608538,0.6935531,0.014915863,0.21190147,0.07330607,0.00018190895],"about_ca_topic_score_codex":0.0073888646,"about_ca_topic_score_gemma":0.007911489,"teacher_disagreement_score":0.010397868,"about_ca_system_score_codex":0.0018298962,"about_ca_system_score_gemma":0.0035579149,"threshold_uncertainty_score":0.054989815},"labels":[],"label_agreement":null},{"id":"W4292946475","doi":"","title":"ntegrating Formal Program Verification with Testing","year":2012,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Prevention of Organ Failure","funders":"","keywords":"Computer science; Programming language; Formal verification; Formal methods; Software engineering","score_opus":0.028285916040893998,"score_gpt":0.2502259271547433,"score_spread":0.22194001111384928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4292946475","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0065510315,0.00019523756,0.9866514,0.00032268313,0.00009028292,0.00006574696,0.000028654538,0.0022691,0.003825781],"genre_scores_gemma":[0.50320923,0.0006027443,0.4833412,0.00055640313,0.00023459461,0.0002744124,0.00030972986,0.0015543415,0.009917299],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98635256,0.006892216,0.0007786551,0.0014964743,0.0037628033,0.0007173242],"domain_scores_gemma":[0.96567976,0.024744458,0.0010845123,0.006652278,0.001525161,0.00031381674],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070236614,0.0015244458,0.0014681359,0.0016350866,0.00075806404,0.002124588,0.0024794375,0.0018034556,0.00920589],"category_scores_gemma":[0.03724254,0.0010171884,0.0021939217,0.0009878253,0.0042508333,0.0066926894,0.005038496,0.003672596,0.0019606133],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069857924,0.0003140472,0.0021561766,0.0011496877,0.00016506536,0.00090113474,0.0004994211,0.113150924,0.020231243,0.39336646,0.0044429074,0.4629243],"study_design_scores_gemma":[0.00007798223,0.0001252786,0.00023938032,0.00015824033,0.000064536136,0.00029817648,0.000048695387,0.48911867,0.022032842,0.47979203,0.008009535,0.00003472732],"about_ca_topic_score_codex":0.0013324958,"about_ca_topic_score_gemma":0.0011644468,"teacher_disagreement_score":0.00920589,"about_ca_system_score_codex":0.001227405,"about_ca_system_score_gemma":0.0014139552,"threshold_uncertainty_score":0.037145138},"labels":[],"label_agreement":null},{"id":"W4293870966","doi":"","title":"Third International Competition on Runtime Verification CRV 2016","year":2016,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Computer science; Competition (biology); Runtime verification; Programming language; Formal verification","score_opus":0.01578126948584593,"score_gpt":0.23936021654750028,"score_spread":0.22357894706165435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293870966","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022006191,0.040689394,0.40975127,0.041930653,0.1764762,0.0022195973,0.010420717,0.051745422,0.24476054],"genre_scores_gemma":[0.119902015,0.0150462035,0.1483557,0.0058573633,0.026002128,0.0012457425,0.056250025,0.035917163,0.59142363],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9835079,0.004565334,0.0008736256,0.0026823296,0.006311734,0.0020590664],"domain_scores_gemma":[0.97176456,0.004266067,0.00058368564,0.006945573,0.009956534,0.0064835376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018448718,0.0036464562,0.004547644,0.0044373823,0.0023843518,0.009057382,0.005029366,0.0046821935,0.09920961],"category_scores_gemma":[0.024684003,0.0011909447,0.00369422,0.0036882842,0.0015899,0.0059379386,0.0077507785,0.0051749824,0.06350431],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006728388,0.00023282424,0.0003324209,0.0006571303,0.00010319892,0.0001334205,0.00011841331,0.003731296,0.0044722743,0.017143745,0.72618806,0.24621443],"study_design_scores_gemma":[0.000333237,0.0004161282,0.0014159743,0.0005154945,0.00007707176,0.0003165087,0.00012033624,0.021786982,0.0060444903,0.034224648,0.9346609,0.00008828529],"about_ca_topic_score_codex":0.006198768,"about_ca_topic_score_gemma":0.008275033,"teacher_disagreement_score":0.09920961,"about_ca_system_score_codex":0.004241017,"about_ca_system_score_gemma":0.008392742,"threshold_uncertainty_score":0.3318892},"labels":[],"label_agreement":null},{"id":"W4296938093","doi":"","title":"Validating a dynamic signature monitoring approach using the LTL model checking technique","year":2004,"lang":"en","type":"preprint","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Signature (topology); Computer science; Model checking; Algorithm; Mathematics","score_opus":0.03016117812969912,"score_gpt":0.2755443541943241,"score_spread":0.24538317606462498,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4296938093","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032660488,0.000098679135,0.9508508,0.00035633543,0.00009517578,0.00009512079,0.00017979044,0.013380643,0.0022831005],"genre_scores_gemma":[0.74661136,0.00016117623,0.24655932,0.00045860134,0.00007294913,0.00014193101,0.0003752436,0.0017568521,0.0038625137],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9923908,0.002226693,0.00042152492,0.0011561371,0.003056223,0.0007486954],"domain_scores_gemma":[0.9825968,0.00653388,0.0014016163,0.0071548144,0.0020987906,0.00021401691],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00434942,0.0012309444,0.0010149301,0.0013210746,0.0009991156,0.0026155382,0.002862043,0.002269489,0.0035863174],"category_scores_gemma":[0.015693402,0.00095771725,0.001694721,0.0007975875,0.002052795,0.004552132,0.0026549236,0.0021913303,0.0010001294],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00161045,0.00070877885,0.018163184,0.00088449614,0.00045464074,0.0025573054,0.0010712753,0.29819414,0.23122805,0.14675835,0.011125615,0.28724375],"study_design_scores_gemma":[0.00008602383,0.00015881239,0.0007142752,0.00006673195,0.00014011895,0.0002928427,0.000038201833,0.8309411,0.1331745,0.02863602,0.005698754,0.000052623836],"about_ca_topic_score_codex":0.004491539,"about_ca_topic_score_gemma":0.006427865,"teacher_disagreement_score":0.004491539,"about_ca_system_score_codex":0.0017437041,"about_ca_system_score_gemma":0.0037543469,"threshold_uncertainty_score":0.023002207},"labels":[],"label_agreement":null},{"id":"W4297854797","doi":"","title":"19 - A Comparison of the Specification Methods","year":2006,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Programming language","score_opus":0.04149303930787158,"score_gpt":0.31590510906213515,"score_spread":0.2744120697542636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4297854797","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023896715,0.0020402642,0.940928,0.00087751343,0.00026960103,0.0003045735,0.0007275702,0.0051803743,0.02577556],"genre_scores_gemma":[0.30355275,0.0023005968,0.66935647,0.00042198465,0.000101123485,0.0006141056,0.001900144,0.0027333684,0.019019464],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.97632253,0.008606374,0.0016436488,0.0010595302,0.011414496,0.00095345726],"domain_scores_gemma":[0.9379896,0.03809086,0.0018821686,0.009919131,0.011292483,0.0008256614],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014197577,0.0008411334,0.000892014,0.0026240118,0.0007070128,0.003191324,0.00317694,0.0019019889,0.020220341],"category_scores_gemma":[0.048604816,0.0005410108,0.0013033623,0.0019663011,0.0014051575,0.003722499,0.0021193468,0.00214052,0.0052983197],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021357553,0.0004121566,0.0058513414,0.0017127978,0.00019008422,0.00016129161,0.0011605351,0.022195812,0.009849859,0.23344369,0.013056224,0.70983046],"study_design_scores_gemma":[0.0015150381,0.0021900174,0.014023467,0.0013619885,0.00064185547,0.0015468989,0.002489602,0.47192407,0.10390877,0.13945365,0.26055095,0.0003937455],"about_ca_topic_score_codex":0.0034124528,"about_ca_topic_score_gemma":0.0028772084,"teacher_disagreement_score":0.020220341,"about_ca_system_score_codex":0.0016996945,"about_ca_system_score_gemma":0.0045641377,"threshold_uncertainty_score":0.075084865},"labels":[],"label_agreement":null},{"id":"W4298876240","doi":"","title":"Automatic generation of vulnerability test suite for the Java Card verifier","year":2011,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Test suite; Java Card; Java; Suite; Vulnerability (computing); Operating system; Fuzz testing; Programming language; Computer security; Test case; Software; Real time Java","score_opus":0.04849763494859158,"score_gpt":0.26407298234755955,"score_spread":0.21557534739896797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4298876240","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27102312,0.0004901675,0.6810046,0.00035922573,0.00016925992,0.00036017402,0.001577379,0.041778434,0.0032375841],"genre_scores_gemma":[0.70616573,0.00013331097,0.28663385,0.00012144366,0.000042265794,0.00021268683,0.0028150834,0.0019165691,0.001959009],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982496,0.0005063199,0.00011299679,0.00031789413,0.00059913844,0.00021391595],"domain_scores_gemma":[0.99515355,0.002688761,0.00044686062,0.0007012611,0.0008388095,0.00017069296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011831056,0.001073899,0.000983114,0.0021711262,0.0004027794,0.00093392533,0.0010832471,0.0010614333,0.0037800667],"category_scores_gemma":[0.0059238877,0.0005352076,0.0010013632,0.00058563484,0.00045408972,0.000975068,0.0010278034,0.00083395786,0.0010048073],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015843548,0.0005537235,0.014054455,0.0009758399,0.00027313392,0.002921539,0.00050926575,0.081525445,0.36265036,0.011668409,0.01815689,0.5051266],"study_design_scores_gemma":[0.00032338212,0.00052985473,0.0051120934,0.000084760904,0.0001753798,0.001428928,0.000099884375,0.7510476,0.22427158,0.010143634,0.006693886,0.00008898162],"about_ca_topic_score_codex":0.0011691995,"about_ca_topic_score_gemma":0.0014831397,"teacher_disagreement_score":0.0037800667,"about_ca_system_score_codex":0.00044895,"about_ca_system_score_gemma":0.0010212556,"threshold_uncertainty_score":0.012645602},"labels":[],"label_agreement":null},{"id":"W4300469825","doi":"10.1145/2513228.2513286","title":"An interactive graph-based automation assistant","year":2013,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Component (thermodynamics); Graph; Distributed computing; Interconnection; Computer network; Theoretical computer science","score_opus":0.024337433818536102,"score_gpt":0.3006874084814499,"score_spread":0.27634997466291383,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4300469825","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019238774,0.00008902217,0.77683717,0.00028179734,0.00006244162,0.00025862976,0.000782401,0.18647064,0.015979066],"genre_scores_gemma":[0.2691409,0.0002598918,0.6897833,0.00048199904,0.000082944265,0.00048723718,0.002412504,0.0092173,0.028134009],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994444,0.000107606815,0.00002530608,0.00017348345,0.0001912386,0.000058044872],"domain_scores_gemma":[0.99905306,0.0004935237,0.00006225983,0.00018033477,0.00010352249,0.00010727384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005938448,0.0011757335,0.00042975054,0.0007254629,0.00039608707,0.0013458815,0.0020797737,0.00069035526,0.019203907],"category_scores_gemma":[0.0022083384,0.0004623968,0.0005884063,0.00034039447,0.0004631763,0.0012161464,0.0017810456,0.0009329466,0.0053532906],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021903655,0.0005628115,0.004866936,0.0009839003,0.000118977936,0.0024984647,0.002489599,0.069637895,0.18130197,0.05781465,0.1108469,0.56668746],"study_design_scores_gemma":[0.00031608544,0.00035088518,0.0035661107,0.000090128044,0.00012690214,0.0013398471,0.00042753044,0.51339024,0.12951377,0.024187364,0.32654223,0.00014877411],"about_ca_topic_score_codex":0.0020856773,"about_ca_topic_score_gemma":0.0024953007,"teacher_disagreement_score":0.019203907,"about_ca_system_score_codex":0.000541551,"about_ca_system_score_gemma":0.00074185897,"threshold_uncertainty_score":0.064243436},"labels":[],"label_agreement":null},{"id":"W4300771635","doi":"10.1145/2001420","title":"Proceedings of the 2011 International Symposium on Software Testing and Analysis","year":2011,"lang":"en","type":"paratext","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Software testing; Computer science; Software engineering; Software; Programming language","score_opus":0.03037372862193266,"score_gpt":0.2551291195448247,"score_spread":0.22475539092289204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4300771635","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014267592,0.04573795,0.5299022,0.026338058,0.052096117,0.0011572393,0.003395464,0.009641013,0.31746432],"genre_scores_gemma":[0.08694558,0.049533673,0.23056927,0.0052721226,0.0157822,0.0013123464,0.019408934,0.0059941118,0.5851818],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99201363,0.0023193324,0.00050963863,0.00064138876,0.004046523,0.00046946912],"domain_scores_gemma":[0.98724294,0.0040974175,0.00042508935,0.0024895268,0.0043866993,0.001358278],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00748301,0.0017392087,0.001872684,0.0032619191,0.0012006778,0.007193287,0.0020850066,0.0025438997,0.07197911],"category_scores_gemma":[0.0150000295,0.00078477437,0.0014123417,0.0020399478,0.0016465349,0.004490149,0.002287531,0.0048963665,0.034746394],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028798962,0.00038319392,0.0016863429,0.00048175512,0.00012514523,0.00034542492,0.00035884097,0.00258086,0.004908619,0.026746953,0.49146035,0.4706345],"study_design_scores_gemma":[0.000049329872,0.000247273,0.0021477696,0.0005560025,0.00008319421,0.000869612,0.0001408539,0.008873438,0.004421366,0.028435158,0.95411754,0.00005856592],"about_ca_topic_score_codex":0.0020877526,"about_ca_topic_score_gemma":0.0042741923,"teacher_disagreement_score":0.07197911,"about_ca_system_score_codex":0.00138605,"about_ca_system_score_gemma":0.0032300304,"threshold_uncertainty_score":0.24079412},"labels":[],"label_agreement":null},{"id":"W4300949634","doi":"","title":"Integration testing of communicating systems with unknown components","year":2015,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Integration testing; Computer science; Programming language","score_opus":0.04369631623138119,"score_gpt":0.24079536833187573,"score_spread":0.19709905210049455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4300949634","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.48105547,0.00030971217,0.5135792,0.00021383789,0.00004939489,0.00006128048,0.000044455723,0.002654865,0.0020318618],"genre_scores_gemma":[0.94716346,0.00007229199,0.051818926,0.000041356878,0.0000171563,0.000046748053,0.00006430157,0.00016500951,0.0006107219],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9886719,0.0043893177,0.00052635575,0.0009596092,0.0046362025,0.0008165695],"domain_scores_gemma":[0.9532002,0.036918707,0.0025757053,0.004079989,0.0026246344,0.0006007728],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038503143,0.0012027577,0.00096251466,0.0011990834,0.0005306964,0.001428394,0.0024554175,0.0017116194,0.0014397989],"category_scores_gemma":[0.032221302,0.0007237021,0.00095030526,0.00087604683,0.0020627435,0.0023039137,0.0018095631,0.0015033038,0.0002612349],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002897484,0.0010375191,0.03319909,0.0010298957,0.000614303,0.0045581935,0.0027365063,0.49998462,0.15531327,0.056773502,0.0014781924,0.2403774],"study_design_scores_gemma":[0.00012255002,0.0004903834,0.002627399,0.00007998436,0.00015478351,0.00059430744,0.00010659207,0.91359746,0.059596572,0.021772318,0.00082445104,0.00003315185],"about_ca_topic_score_codex":0.0015916161,"about_ca_topic_score_gemma":0.0011942766,"teacher_disagreement_score":0.0038503143,"about_ca_system_score_codex":0.0008252151,"about_ca_system_score_gemma":0.0011879066,"threshold_uncertainty_score":0.020362616},"labels":[],"label_agreement":null},{"id":"W4308627374","doi":"10.1145/3558489.3559073","title":"On the effectiveness of data balancing techniques in the context of ML-based test case prioritization","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Regression testing; Computer science; Context (archaeology); Prioritization; Regression analysis; Regression; Machine learning; Test data; Risk-based testing; Data mining; Reliability engineering; Artificial intelligence; Software; Software system; Statistics; Engineering; Software engineering","score_opus":0.034384400689239744,"score_gpt":0.2944197283149523,"score_spread":0.2600353276257125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4308627374","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7286304,0.01888174,0.23103158,0.0026410604,0.00061903725,0.00095082837,0.0028487674,0.006968753,0.0074277353],"genre_scores_gemma":[0.87604195,0.0011829496,0.11803695,0.00031345765,0.00027508114,0.00025390624,0.0031281426,0.00022739747,0.0005401588],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9754951,0.011894841,0.0024432254,0.0038192933,0.0057006846,0.0006469354],"domain_scores_gemma":[0.84663117,0.121938676,0.008908974,0.013797609,0.0074255294,0.001298009],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02384683,0.0019267062,0.0011262397,0.0062390887,0.0009782908,0.0029670224,0.0020541092,0.0015167458,0.0010769562],"category_scores_gemma":[0.122443326,0.00035755048,0.0010981074,0.0036976053,0.001126362,0.0047546495,0.0016875138,0.0021782285,0.0007627266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019807245,0.0016282293,0.11148527,0.0015587324,0.00090505375,0.00019144535,0.00070892734,0.07993153,0.019863665,0.002826675,0.006490363,0.77242935],"study_design_scores_gemma":[0.0004437495,0.004121051,0.10391059,0.000848962,0.000944414,0.000972589,0.0019404877,0.8073924,0.05061587,0.012325166,0.016303489,0.00018117846],"about_ca_topic_score_codex":0.002049648,"about_ca_topic_score_gemma":0.002312161,"teacher_disagreement_score":0.02384683,"about_ca_system_score_codex":0.00091911096,"about_ca_system_score_gemma":0.0017294771,"threshold_uncertainty_score":0.12611562},"labels":[],"label_agreement":null},{"id":"W4309874955","doi":"10.1016/j.jss.2022.111543","title":"Configuring mission-specific behavior in a product line of collaborating Small Unmanned Aerial Systems","year":2022,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Drone; Task (project management); Process (computing); Set (abstract data type); Computer science; Systems engineering; Software; Real-time computing; Distributed computing; Software engineering; Reliability engineering; Engineering; Operating system","score_opus":0.03856794910153191,"score_gpt":0.2618310104852299,"score_spread":0.22326306138369797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4309874955","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9737627,0.000021849219,0.024834054,0.000028380999,0.0000047110734,0.00004411635,0.000030375453,0.0006120021,0.00066178286],"genre_scores_gemma":[0.9864119,0.000008530915,0.013059959,0.000008209517,9.221159e-7,0.000014611589,0.000057703164,0.000048798098,0.00038921623],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918956,0.00025697367,0.000049275885,0.00018968263,0.0002338409,0.00008062487],"domain_scores_gemma":[0.9948096,0.0022200684,0.00066437334,0.0011284279,0.0008612376,0.00031627945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011232884,0.0005426469,0.00021130714,0.000491229,0.00042864666,0.0005376301,0.0007533507,0.00052791124,0.0010502477],"category_scores_gemma":[0.0051397425,0.0003120661,0.00021735326,0.0002157364,0.00045724926,0.0006738318,0.00047684097,0.00049904507,0.00018438733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018566243,0.0017076025,0.10285543,0.00021027769,0.00014083918,0.0032015855,0.004203022,0.35817698,0.23968509,0.0026529485,0.001715479,0.28359404],"study_design_scores_gemma":[0.00004944306,0.0012762282,0.022177665,0.000019703642,0.0000514749,0.00040935652,0.0005248663,0.9257386,0.04745697,0.0010807008,0.0011826532,0.000032221575],"about_ca_topic_score_codex":0.0033377095,"about_ca_topic_score_gemma":0.0038102255,"teacher_disagreement_score":0.0033377095,"about_ca_system_score_codex":0.00051203725,"about_ca_system_score_gemma":0.00044726641,"threshold_uncertainty_score":0.00663656},"labels":[],"label_agreement":null},{"id":"W4310506204","doi":"10.1145/3563767.3568132","title":"Evaluating the Quality of Student-Written Software Tests with Curated Mutation Analysis","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Test suite; Correctness; Programming language; Assertion; Software engineering; Software quality; Code coverage; Software; Software testing; Test (biology); Invocation; Test case; Software development; Machine learning","score_opus":0.1151691530031579,"score_gpt":0.42743782641175704,"score_spread":0.31226867340859915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4310506204","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9448762,0.00045813454,0.048443623,0.00013227726,0.000080654594,0.00015636867,0.00060865004,0.0024473977,0.002796732],"genre_scores_gemma":[0.95933264,0.00013143607,0.036408976,0.000078241705,0.000019881176,0.000081918515,0.001719666,0.0004891802,0.0017380546],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9876993,0.0038362592,0.0013918,0.001500375,0.005110151,0.0004620425],"domain_scores_gemma":[0.87657243,0.07201292,0.012684359,0.008971444,0.027636549,0.0021221892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008438538,0.0008531847,0.00066803134,0.002473543,0.00026177894,0.001624373,0.0011876283,0.0012745126,0.0018857617],"category_scores_gemma":[0.10193114,0.00025893422,0.00059814774,0.0014453242,0.0006351219,0.0011530602,0.0011855887,0.0007118813,0.00085211877],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0062888833,0.0037931956,0.25375885,0.0012106167,0.001677648,0.0010264376,0.0023917793,0.14072996,0.07871723,0.0015784886,0.0057854992,0.5030414],"study_design_scores_gemma":[0.00067304546,0.012275443,0.32560202,0.00046233324,0.00065006904,0.0016644498,0.0010809164,0.47536385,0.16959691,0.0029626205,0.009329629,0.00033872735],"about_ca_topic_score_codex":0.0020995422,"about_ca_topic_score_gemma":0.0027850168,"teacher_disagreement_score":0.008438538,"about_ca_system_score_codex":0.0006871691,"about_ca_system_score_gemma":0.0009029175,"threshold_uncertainty_score":0.044627786},"labels":[],"label_agreement":null},{"id":"W4310673963","doi":"10.1016/j.infsof.2022.107129","title":"A probabilistic framework for mutation testing in deep neural networks","year":2022,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Testability; Context (archaeology); Machine learning; Probabilistic logic; Mutation; Set (abstract data type); Artificial intelligence; Test suite; Artificial neural network; Suite; Data mining; Model-based testing; Reliability engineering; Test case; Engineering","score_opus":0.01667141722285493,"score_gpt":0.2536663021430027,"score_spread":0.23699488492014775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4310673963","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0041377693,0.00017332425,0.9943461,0.00017938553,0.000019676756,0.00002385283,0.00006151788,0.00047607586,0.00058229855],"genre_scores_gemma":[0.5691989,0.000492399,0.42354137,0.0003351642,0.000183218,0.00035882916,0.00042999344,0.0006159001,0.0048442725],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956766,0.0016777877,0.0002505497,0.0006490391,0.0012698577,0.00047618558],"domain_scores_gemma":[0.98187226,0.013831511,0.0009904505,0.0011498329,0.0017649679,0.00039090213],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006901001,0.0012055277,0.0019505036,0.002513807,0.00083725236,0.001989259,0.0050488743,0.0026927092,0.0045133266],"category_scores_gemma":[0.028350491,0.0013508847,0.0018454047,0.001757536,0.0029071206,0.004431457,0.003290173,0.0030640007,0.0004295779],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009238568,0.000053136228,0.0007586619,0.00009121014,0.00006338503,0.00009305923,0.00006287444,0.84624165,0.00075881195,0.1078822,0.0011878473,0.04271475],"study_design_scores_gemma":[0.0000045801844,0.0000058675996,0.000042574615,0.000008895208,0.0000056317567,0.0000095640935,0.0000024902054,0.9663716,0.0001681304,0.03324368,0.00013239188,0.0000046324817],"about_ca_topic_score_codex":0.011545774,"about_ca_topic_score_gemma":0.016298143,"teacher_disagreement_score":0.011545774,"about_ca_system_score_codex":0.0029033138,"about_ca_system_score_gemma":0.0030923898,"threshold_uncertainty_score":0.0364964},"labels":[],"label_agreement":null},{"id":"W4312260323","doi":"10.1109/tse.2022.3217544","title":"Dynamic Human-in-the-Loop Assertion Generation","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of British Columbia","funders":"","keywords":"Assertion; Computer science; Programming language; TypeScript; JavaScript; Test (biology); Test case; Workflow; Notation; Automation; Software engineering; Variable (mathematics); Database; Arithmetic; Mathematics; Machine learning","score_opus":0.02170916402304557,"score_gpt":0.2527178957400945,"score_spread":0.23100873171704894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312260323","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08029779,0.00014883779,0.82911384,0.00025825226,0.00019483402,0.00090489525,0.0017700059,0.08249565,0.004815862],"genre_scores_gemma":[0.36279193,0.00013704173,0.61447996,0.00035420002,0.00006843639,0.0011454072,0.0052591586,0.010457136,0.0053067775],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99501747,0.0014787145,0.00040174296,0.001361619,0.001425934,0.00031457323],"domain_scores_gemma":[0.96159846,0.026729865,0.0019503839,0.0050847474,0.0041812467,0.00045528225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0063063125,0.0013946992,0.00054005353,0.0016744942,0.0003932909,0.0013510083,0.0019337161,0.0007406588,0.006553347],"category_scores_gemma":[0.042927235,0.00075302913,0.0008070478,0.0005642779,0.0009706139,0.0015971626,0.0017751135,0.0012127737,0.0029760385],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012664472,0.0011484748,0.04237473,0.0016338762,0.00022429395,0.003942758,0.0057012117,0.06717591,0.08944133,0.029774154,0.05652895,0.7007878],"study_design_scores_gemma":[0.00035865276,0.0005776924,0.0065827863,0.0003688415,0.00013320209,0.0018594358,0.00060668547,0.6834618,0.19675201,0.026169928,0.08294416,0.00018474436],"about_ca_topic_score_codex":0.0011430167,"about_ca_topic_score_gemma":0.0013313796,"teacher_disagreement_score":0.006553347,"about_ca_system_score_codex":0.00048670958,"about_ca_system_score_gemma":0.0014916699,"threshold_uncertainty_score":0.033351302},"labels":[],"label_agreement":null},{"id":"W4312349853","doi":"10.1145/3524610.3527902","title":"XAI4FL","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Statement (logic); Computer science; Fault (geology); Ranking (information retrieval); Test (biology); Unit testing; Class (philosophy); Software bug; Problem statement; Test case; Software; Programming language; Artificial intelligence; Machine learning; Engineering","score_opus":0.013764278414417614,"score_gpt":0.22550378973091126,"score_spread":0.21173951131649366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312349853","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062037287,0.0014017439,0.21035379,0.0013258617,0.0008733338,0.0004766324,0.029109253,0.59440446,0.15585129],"genre_scores_gemma":[0.095288485,0.0021825698,0.2956871,0.0024392123,0.000495248,0.0014127467,0.16888869,0.0969927,0.33661327],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986738,0.00020638583,0.00009274121,0.0003408547,0.000494822,0.00019146729],"domain_scores_gemma":[0.9978346,0.0004374335,0.000128623,0.0009033604,0.00057451974,0.00012146927],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002317751,0.0020698141,0.0007385426,0.0023529676,0.0011121321,0.0029776942,0.004238141,0.0017315346,0.19251801],"category_scores_gemma":[0.004134164,0.0011965887,0.0010828907,0.0013473963,0.0006342772,0.0035613887,0.002585119,0.002238097,0.10618022],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010234156,0.00015443281,0.0021740696,0.0011423539,0.0001114353,0.000240647,0.00021756759,0.0026206998,0.009976429,0.026260668,0.6835879,0.27249038],"study_design_scores_gemma":[0.00020462816,0.0001664958,0.0014484642,0.00015107816,0.000047800528,0.0003447663,0.000050251278,0.01667717,0.01844445,0.009812018,0.9525693,0.00008352611],"about_ca_topic_score_codex":0.0033367833,"about_ca_topic_score_gemma":0.0045022657,"teacher_disagreement_score":0.19251801,"about_ca_system_score_codex":0.0012018421,"about_ca_system_score_gemma":0.0012761852,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4312429593","doi":"10.1109/icsme55016.2022.00039","title":"What Made This Test Flake? Pinpointing Classes Responsible for Test Flakiness","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Computer science; Regression testing; Code (set theory); Quality (philosophy); Fault (geology); Test (biology); Test suite; Class (philosophy); Reliability engineering; Debugging; Test case; Programming language; Software; Machine learning; Artificial intelligence; Software development; Engineering; Regression analysis","score_opus":0.030598165685785494,"score_gpt":0.2858525494159046,"score_spread":0.2552543837301191,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312429593","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.65746474,0.003376654,0.2677212,0.0056601646,0.0009772391,0.00048656887,0.0015267024,0.055431187,0.007355538],"genre_scores_gemma":[0.89634234,0.0003678418,0.09640236,0.0009811715,0.00009341378,0.000099780336,0.00091603806,0.0020624886,0.0027345368],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99383,0.0013092875,0.00037870952,0.0015472758,0.0021888602,0.00074590073],"domain_scores_gemma":[0.9647248,0.018030044,0.0061321333,0.0048629325,0.0042725415,0.00197764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033004968,0.0016882938,0.00087145285,0.003968749,0.00079443847,0.00231046,0.0016317617,0.001957397,0.0029836188],"category_scores_gemma":[0.048706952,0.00064852036,0.0011655932,0.0015853567,0.0013679672,0.0036347087,0.0016304882,0.0022673479,0.0019538593],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014233593,0.0004170549,0.17250928,0.0011565844,0.0002998816,0.0034016878,0.0031772056,0.018547276,0.13050434,0.005081101,0.01978953,0.6436927],"study_design_scores_gemma":[0.0003699691,0.0024288248,0.19870767,0.0015217103,0.0011776716,0.008186671,0.0048550027,0.36303827,0.32320172,0.02094549,0.074989356,0.0005777408],"about_ca_topic_score_codex":0.0057978276,"about_ca_topic_score_gemma":0.008181059,"teacher_disagreement_score":0.0057978276,"about_ca_system_score_codex":0.0011690558,"about_ca_system_score_gemma":0.0019727836,"threshold_uncertainty_score":0.017454922},"labels":[],"label_agreement":null},{"id":"W4312529070","doi":"10.1145/3510457.3513038","title":"The impact of flaky tests on historical test prioritization on chrome","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Prioritization; Blocking (statistics); Pipeline (software); Computer science; Test (biology); Reliability engineering; Work (physics); Code (set theory); Regression testing; Engineering; Software; Software development; Operating system; Programming language; Computer network","score_opus":0.019321535303824963,"score_gpt":0.28688465220610365,"score_spread":0.2675631169022787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312529070","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80592036,0.005013889,0.12450892,0.0025597257,0.00056464516,0.00040871426,0.0014164919,0.04578007,0.013827192],"genre_scores_gemma":[0.8910775,0.0003492547,0.10325674,0.0005601422,0.000048554808,0.00006085945,0.0014336727,0.0012779074,0.0019354009],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9875145,0.0031809707,0.0008140423,0.0027352376,0.0048676347,0.00088760833],"domain_scores_gemma":[0.9087146,0.058856215,0.0044736406,0.014809388,0.011038991,0.002107217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011015083,0.0014260515,0.0008197294,0.002837647,0.0009566762,0.003081822,0.0025532586,0.0013366726,0.0017262684],"category_scores_gemma":[0.084149934,0.0007158888,0.00071131816,0.0018485385,0.0016862442,0.0042836806,0.0014916033,0.002367216,0.00069963554],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024600634,0.0010965762,0.1388202,0.0006630106,0.00038602666,0.00037843065,0.00058993086,0.1897071,0.028408807,0.0062226164,0.022489348,0.6087779],"study_design_scores_gemma":[0.0003255453,0.001509262,0.050260305,0.00016627488,0.00021720423,0.0006474115,0.0004278643,0.8827317,0.046442337,0.0052968664,0.011792735,0.00018260992],"about_ca_topic_score_codex":0.02494034,"about_ca_topic_score_gemma":0.030121945,"teacher_disagreement_score":0.02494034,"about_ca_system_score_codex":0.002679256,"about_ca_system_score_gemma":0.0028492429,"threshold_uncertainty_score":0.058254063},"labels":[],"label_agreement":null},{"id":"W4312663631","doi":"10.1007/978-3-031-19762-8_14","title":"Model-Driven Engineering in Digital Thread Platforms: A Practical Use Case and Future Challenges","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"University of Galway; Science Foundation Ireland; European Commission; National University of Ireland","keywords":"Computer science; Interoperability; Software engineering; Model-driven architecture; Thread (computing); Orchestration; Context (archaeology); Software; Distributed computing; Software development; World Wide Web; Programming language","score_opus":0.05086537225321081,"score_gpt":0.2692703214509965,"score_spread":0.2184049491977857,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312663631","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13254121,0.0028748044,0.74580705,0.008368544,0.00039048394,0.00033429274,0.00021740275,0.0032923887,0.106173806],"genre_scores_gemma":[0.4799077,0.0023052807,0.475238,0.0005520145,0.000120446195,0.00026294834,0.00030530075,0.000849443,0.040458955],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986507,0.0005500158,0.00006454725,0.0001093705,0.0005180987,0.000107310945],"domain_scores_gemma":[0.99840146,0.00089578895,0.00006057435,0.00035456874,0.00017408974,0.000113426664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002594954,0.00047425786,0.00039345742,0.0005145025,0.0009954382,0.0040427037,0.0014928387,0.0022485466,0.00516323],"category_scores_gemma":[0.0032401648,0.00031671216,0.0005613065,0.00076334574,0.0014309393,0.0036464927,0.0024160142,0.0020648441,0.0011094746],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014161826,0.0004622647,0.0018669927,0.0005107477,0.000029774252,0.0034164246,0.002581856,0.10670439,0.013786202,0.6831938,0.012865217,0.17444067],"study_design_scores_gemma":[0.000091299386,0.00020004812,0.00043692687,0.000387639,0.00003200872,0.0013784134,0.0008591482,0.39515814,0.017906008,0.24430732,0.3391729,0.000070185495],"about_ca_topic_score_codex":0.0021312635,"about_ca_topic_score_gemma":0.0021695597,"teacher_disagreement_score":0.00516323,"about_ca_system_score_codex":0.0013633384,"about_ca_system_score_gemma":0.0012552512,"threshold_uncertainty_score":0.01727271},"labels":[],"label_agreement":null},{"id":"W4312751706","doi":"10.1007/978-3-031-19756-7_17","title":"On Technical Debt in Software Testing - Observations from Industry","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Technical debt; Computer science; Agile software development; Code refactoring; Test suite; Test case; Software engineering; Test Management Approach; White-box testing; Code coverage; Automation; Test (biology); Software; Software system; Software development; Programming language; Software construction; Engineering; Machine learning","score_opus":0.05620035206899127,"score_gpt":0.2735181785792026,"score_spread":0.2173178265102113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312751706","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7979055,0.017840393,0.006292589,0.023184087,0.00010597439,0.000037000704,0.00077608455,0.00024687906,0.15361148],"genre_scores_gemma":[0.9911466,0.0025685725,0.0005782207,0.0007073248,0.00006740126,0.0000086662585,0.00026939635,0.000047772413,0.0046060844],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99852175,0.0004761654,0.00009571163,0.00014257591,0.0005422671,0.00022148796],"domain_scores_gemma":[0.94548386,0.043558322,0.0043036025,0.0017837366,0.003658538,0.0012118892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002966444,0.00018541659,0.00024693698,0.002178135,0.0010115684,0.0018557681,0.0007690266,0.0013238827,0.004506756],"category_scores_gemma":[0.030456167,0.00019935658,0.00019673431,0.0044259736,0.00175365,0.004709128,0.0014903474,0.0021711923,0.00057286455],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009148119,0.0007094489,0.33293748,0.00052134326,0.00003752834,0.0031165185,0.025882384,0.00499277,0.0022158558,0.13321891,0.059396517,0.43605652],"study_design_scores_gemma":[0.0000897304,0.00047516025,0.6215741,0.0013524778,0.00006364465,0.0028008895,0.034299728,0.01132005,0.0046965764,0.16019174,0.1630291,0.000106731204],"about_ca_topic_score_codex":0.011025825,"about_ca_topic_score_gemma":0.011526373,"teacher_disagreement_score":0.011025825,"about_ca_system_score_codex":0.0014477706,"about_ca_system_score_gemma":0.0008126581,"threshold_uncertainty_score":0.021923304},"labels":[],"label_agreement":null},{"id":"W4312785754","doi":"10.1145/3510454.3516840","title":"MASS","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Horizon 2020 Framework Programme; European Commission","keywords":"Computer science; Software; Cyber-physical system; Operating system","score_opus":0.015569225563169567,"score_gpt":0.23125383672071712,"score_spread":0.21568461115754756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312785754","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029366896,0.0016800901,0.6335116,0.0008898498,0.00057768443,0.0004301256,0.006682093,0.2093008,0.11756077],"genre_scores_gemma":[0.37878102,0.0015902123,0.4686745,0.0015915368,0.0003959235,0.00069401134,0.013310905,0.0225032,0.112458736],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991304,0.000110826964,0.000045559096,0.00021397465,0.00041643056,0.000082798106],"domain_scores_gemma":[0.99881446,0.0004759102,0.00012618756,0.0002822902,0.0002489357,0.0000521247],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071816426,0.00122616,0.00058555586,0.0016741595,0.0006334625,0.0015821853,0.0012149371,0.00086113566,0.048288517],"category_scores_gemma":[0.0031814142,0.00044049372,0.00077270385,0.0005935752,0.0005023446,0.0023137191,0.0015172253,0.0008076817,0.01328554],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070208637,0.00021882703,0.0059454115,0.0010859363,0.00017165842,0.0012802306,0.0005194788,0.013429925,0.03158083,0.08716786,0.16119987,0.6966979],"study_design_scores_gemma":[0.00027256002,0.00048500742,0.0038149953,0.0004285755,0.00020627191,0.0029229235,0.00018761292,0.16562477,0.080104165,0.07177325,0.6740618,0.00011810017],"about_ca_topic_score_codex":0.0009981156,"about_ca_topic_score_gemma":0.0015921731,"teacher_disagreement_score":0.048288517,"about_ca_system_score_codex":0.00040143757,"about_ca_system_score_gemma":0.0006170233,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4312796785","doi":"10.1145/3510454.3516856","title":"DScribe","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Documentation; Computer science; Redundancy (engineering); Unit testing; Artifact (error); Technical documentation; Template; Software engineering; Programming language; Software; Operating system; Artificial intelligence","score_opus":0.014937335638861552,"score_gpt":0.22409066610293493,"score_spread":0.2091533304640734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312796785","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038281046,0.0011482008,0.3186593,0.001349788,0.0007746414,0.0009750488,0.023231065,0.43039867,0.21963516],"genre_scores_gemma":[0.071559116,0.002242691,0.34299517,0.0034528957,0.00038363328,0.001554914,0.11909704,0.15117568,0.30753887],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9975751,0.00041028042,0.000196423,0.00037286905,0.0012401536,0.00020524011],"domain_scores_gemma":[0.9928871,0.0020098982,0.00029976782,0.002723698,0.0017326794,0.0003467929],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027514293,0.0013405776,0.0007786234,0.0027370723,0.00068650406,0.0036059425,0.004095553,0.0020696928,0.15443216],"category_scores_gemma":[0.018417712,0.0011485738,0.0012359899,0.0012813592,0.00091407925,0.0038506342,0.004442985,0.002670042,0.09675406],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004091099,0.0002056101,0.0015207808,0.0009814706,0.00007377671,0.0004757669,0.00040428404,0.0027686222,0.007769889,0.042690597,0.64254826,0.30015185],"study_design_scores_gemma":[0.00012655153,0.0000738315,0.00086993363,0.00019565538,0.0000204126,0.0006784199,0.000060669325,0.008734526,0.009719303,0.014304465,0.9651492,0.000067082685],"about_ca_topic_score_codex":0.0019505881,"about_ca_topic_score_gemma":0.0034349277,"teacher_disagreement_score":0.15443216,"about_ca_system_score_codex":0.0011631768,"about_ca_system_score_gemma":0.0016462784,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4312834255","doi":"10.1109/tse.2022.3213041","title":"Data-Driven Mutation Analysis for Cyber-Physical Systems","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission; European Space Agency","keywords":"Computer science; Interoperability; Test suite; Set (abstract data type); Data mining; Suite; Mutation; Software; Quality (philosophy); Programming language; Software engineering; Theoretical computer science; Test case; Machine learning; World Wide Web","score_opus":0.03462965960977726,"score_gpt":0.27028679770145464,"score_spread":0.23565713809167738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312834255","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055836827,0.00019343795,0.93877953,0.0002169963,0.000043329695,0.00010649741,0.00026833007,0.0031812924,0.001373741],"genre_scores_gemma":[0.6022233,0.00023985197,0.39436474,0.00017519435,0.00003158809,0.00029629093,0.00086492335,0.00039899402,0.001405177],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997514,0.000590904,0.00014869292,0.0003800354,0.0012150757,0.00015141409],"domain_scores_gemma":[0.9933924,0.004318061,0.0006096191,0.000463956,0.0010688425,0.00014704758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018785819,0.001152268,0.00076801993,0.0037234523,0.0004882949,0.0012552721,0.001282984,0.000888247,0.0012705765],"category_scores_gemma":[0.009719999,0.00033946632,0.00168139,0.0009881011,0.0014551114,0.0012150812,0.0011102479,0.001311142,0.00025580855],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021533576,0.00024923676,0.008937936,0.00031762375,0.00016243673,0.00092630525,0.00023957509,0.733138,0.04398169,0.061809678,0.0017691959,0.14825292],"study_design_scores_gemma":[0.000016686372,0.000044299424,0.0005051692,0.000013784274,0.000015453505,0.00009733525,0.000016505406,0.9732617,0.012465496,0.012771644,0.0007725784,0.000019455392],"about_ca_topic_score_codex":0.0041515785,"about_ca_topic_score_gemma":0.0024889081,"teacher_disagreement_score":0.0041515785,"about_ca_system_score_codex":0.0014708437,"about_ca_system_score_gemma":0.0016988365,"threshold_uncertainty_score":0.010671794},"labels":[],"label_agreement":null},{"id":"W4313119435","doi":"10.1145/3568364.3568380","title":"Identifying Candidate Classes for Unit Testing Using Deep Learning Classifiers: An Empirical Validation","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning; Unit testing; Empirical research; Unit (ring theory); Deep learning; Statistics; Mathematics","score_opus":0.21395416272102558,"score_gpt":0.3930138495221044,"score_spread":0.1790596868010788,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313119435","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9264071,0.00141013,0.0674291,0.0004713914,0.00011179589,0.00022549409,0.0009794139,0.0013153111,0.0016503936],"genre_scores_gemma":[0.95404375,0.00015415195,0.04244515,0.00014456011,0.000027476559,0.0001635165,0.0023188777,0.000072913645,0.0006296642],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9936051,0.0029229347,0.00057097746,0.0012938792,0.0011499574,0.00045711116],"domain_scores_gemma":[0.95146453,0.036441494,0.0017286504,0.00443384,0.0052941767,0.0006372895],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01089139,0.0020517167,0.0009736588,0.002126192,0.00062408415,0.0014395596,0.0022454571,0.0022608312,0.0011367806],"category_scores_gemma":[0.041771304,0.00048353992,0.000990321,0.001086997,0.0010438851,0.0022177747,0.0016102599,0.0024089909,0.00069940416],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002599102,0.0032608332,0.20345601,0.0009897392,0.00069117633,0.00043880666,0.00069307204,0.35935834,0.013479104,0.0024039426,0.012667519,0.39996237],"study_design_scores_gemma":[0.00007608844,0.0004503792,0.009972617,0.0001529195,0.00007832345,0.0000857489,0.0001871282,0.9756246,0.010908749,0.0013923075,0.0010481062,0.000023052937],"about_ca_topic_score_codex":0.004352107,"about_ca_topic_score_gemma":0.0063249194,"teacher_disagreement_score":0.01089139,"about_ca_system_score_codex":0.0012929129,"about_ca_system_score_gemma":0.0015754274,"threshold_uncertainty_score":0.057599902},"labels":[],"label_agreement":null},{"id":"W4315606067","doi":"10.1145/3571253","title":"Taking Back Control in an Intermediate Representation for GPU Computing","year":2023,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada)","funders":"Engineering and Physical Sciences Research Council","keywords":"Computer science; Compiler; Fuzz testing; Programming language; Control flow; Software engineering; Formal specification; Formal methods; Process (computing); Set (abstract data type); Software","score_opus":0.04405870110884504,"score_gpt":0.346137627226172,"score_spread":0.30207892611732695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4315606067","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006005252,0.0000757099,0.987677,0.000372062,0.00007266868,0.000059721602,0.00006549759,0.0021982247,0.0034739692],"genre_scores_gemma":[0.2206352,0.00020909392,0.77186203,0.0005081755,0.000042943662,0.00023460548,0.00031154632,0.0014713285,0.004725119],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99685043,0.0012210398,0.0002647507,0.00029554998,0.0010392382,0.00032897736],"domain_scores_gemma":[0.99537295,0.0023015232,0.00032834068,0.0013908424,0.0005103572,0.00009605968],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003928001,0.0006849745,0.00038667113,0.00082594674,0.00092997396,0.0037024003,0.0020730984,0.0013646809,0.004119326],"category_scores_gemma":[0.011822732,0.0005327427,0.0013510198,0.0005497435,0.0034763918,0.0042908886,0.0026120362,0.0029359218,0.0008271773],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012096757,0.00006382389,0.0011714135,0.0002749314,0.000024496634,0.00035773107,0.0011568791,0.04162013,0.007960535,0.87988424,0.0043820348,0.06298281],"study_design_scores_gemma":[0.000062929146,0.00016196529,0.0003556065,0.00042794447,0.00006342863,0.000438081,0.0003412826,0.27233148,0.03959531,0.533405,0.1527291,0.000087902175],"about_ca_topic_score_codex":0.0032683269,"about_ca_topic_score_gemma":0.0049684956,"teacher_disagreement_score":0.004119326,"about_ca_system_score_codex":0.0019631353,"about_ca_system_score_gemma":0.0029354668,"threshold_uncertainty_score":0.02077353},"labels":[],"label_agreement":null},{"id":"W4319296048","doi":"10.1145/3579640","title":"<scp>Katana</scp> : Dual Slicing Based Context for Learning Bug Fixes","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Program slicing; Debugging; Leverage (statistics); Slicing; Context (archaeology); Program comprehension; Software; Statement (logic); Software engineering; Programming language; Artificial intelligence; Software system; World Wide Web","score_opus":0.0980160634797238,"score_gpt":0.31874940585467515,"score_spread":0.22073334237495135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319296048","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04069626,0.0020716668,0.87386256,0.002125311,0.00040482296,0.00022349572,0.0043118256,0.071790166,0.00451392],"genre_scores_gemma":[0.34204182,0.0011467064,0.63339144,0.0011374474,0.0003060734,0.00036991911,0.011309786,0.0042566443,0.0060401806],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991762,0.00018092083,0.0000472297,0.0002851449,0.00025202043,0.00005841157],"domain_scores_gemma":[0.9971419,0.0010551257,0.0002836091,0.0009223632,0.00045186945,0.00014518568],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00084606046,0.0013433011,0.0005894408,0.001503031,0.00062282593,0.00090414606,0.0018718535,0.001400503,0.005967494],"category_scores_gemma":[0.007060108,0.0005835134,0.00083364965,0.00131787,0.00086790987,0.0024818233,0.0018622386,0.002096198,0.0023915388],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006950329,0.00027714897,0.008787753,0.00073253375,0.0002993096,0.00047040277,0.0004400311,0.08547427,0.035307363,0.00952936,0.13832362,0.71966314],"study_design_scores_gemma":[0.00009130482,0.00028570785,0.0052410127,0.000109537155,0.00011506819,0.00043511434,0.000089190595,0.8986032,0.036872316,0.024046592,0.03401493,0.000096050884],"about_ca_topic_score_codex":0.012689078,"about_ca_topic_score_gemma":0.024255028,"teacher_disagreement_score":0.012689078,"about_ca_system_score_codex":0.0007179716,"about_ca_system_score_gemma":0.0011419692,"threshold_uncertainty_score":0.025230467},"labels":[],"label_agreement":null},{"id":"W4319594569","doi":"10.1145/3583566","title":"COMET: Coverage-guided Model Generation For Deep Learning Library Testing","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Comet; Computer science; Layer (electronics); Set (abstract data type); Test set; Artificial intelligence; Machine learning; Algorithm; Data mining; Programming language; Chemistry","score_opus":0.21094581494621803,"score_gpt":0.3372706176748377,"score_spread":0.12632480272861965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319594569","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12574884,0.0012233091,0.80798423,0.0009854222,0.00014501235,0.00036680378,0.0026354545,0.054515738,0.0063951467],"genre_scores_gemma":[0.6080656,0.0003630286,0.37744516,0.0008249438,0.000037304002,0.0005813542,0.007109942,0.003129092,0.0024435895],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978654,0.0006125343,0.00016461495,0.0003964342,0.0007428766,0.00021814884],"domain_scores_gemma":[0.99350834,0.0040619373,0.00038195815,0.0010908467,0.00079067,0.00016623904],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001818279,0.0021134617,0.00073282106,0.0016081937,0.00048621465,0.001281762,0.003289008,0.0016291079,0.0043526115],"category_scores_gemma":[0.013057856,0.00095710106,0.0020959089,0.0007716082,0.0011127113,0.0024396693,0.0019710774,0.0019334106,0.0010597672],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005086184,0.00040969087,0.015528831,0.0007710829,0.00026011438,0.0007387292,0.0002712176,0.68451536,0.020827383,0.012064461,0.01845272,0.24565172],"study_design_scores_gemma":[0.00004323537,0.00007853045,0.00027136694,0.000022636254,0.000023622893,0.00008396732,0.000021826156,0.98379576,0.008532496,0.0051069534,0.0020077345,0.0000119321885],"about_ca_topic_score_codex":0.008751678,"about_ca_topic_score_gemma":0.013676884,"teacher_disagreement_score":0.008751678,"about_ca_system_score_codex":0.0019704548,"about_ca_system_score_gemma":0.0033161256,"threshold_uncertainty_score":0.017401516},"labels":[],"label_agreement":null},{"id":"W4322489317","doi":"10.1002/stvr.1842","title":"An investigation of distributed computing for combinatorial testing","year":2023,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hypergraph; Computer science; Graph; Theoretical computer science; Vertex (graph theory); Algorithm; Parallel computing; Mathematics; Discrete mathematics","score_opus":0.0456312420782707,"score_gpt":0.2955049233390974,"score_spread":0.2498736812608267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4322489317","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12951382,0.0007234374,0.8289721,0.0026559946,0.00018012563,0.00017564875,0.000052254494,0.0012969696,0.036429748],"genre_scores_gemma":[0.81050545,0.00022717204,0.18516716,0.00020965365,0.00006151509,0.0001357649,0.00005544117,0.00017420294,0.0034636545],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99726164,0.0012834831,0.00007884083,0.00041636798,0.0006438791,0.00031577653],"domain_scores_gemma":[0.9887156,0.007150509,0.00038894126,0.0023477983,0.0010440721,0.00035309844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024632916,0.00049679633,0.0006236676,0.0007277483,0.00086924405,0.0020726672,0.0018419991,0.00078613026,0.0045548202],"category_scores_gemma":[0.008252601,0.0003066521,0.00070264266,0.0010873507,0.0019665277,0.0025683895,0.0015724255,0.0017156082,0.00045314708],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037416277,0.0003156014,0.0035396584,0.00017920393,0.00007288937,0.00041123494,0.00034032736,0.3696015,0.012492608,0.43137625,0.0040332074,0.17726345],"study_design_scores_gemma":[0.00005948031,0.00011484345,0.00043104732,0.00002646798,0.000024888113,0.00013967913,0.00009985949,0.8714729,0.004807635,0.117901094,0.0049089883,0.0000131613615],"about_ca_topic_score_codex":0.0033074613,"about_ca_topic_score_gemma":0.0027534072,"teacher_disagreement_score":0.0045548202,"about_ca_system_score_codex":0.0022579837,"about_ca_system_score_gemma":0.0014880402,"threshold_uncertainty_score":0.016382933},"labels":[],"label_agreement":null},{"id":"W4322754178","doi":"10.1016/j.jss.2023.111666","title":"An empirical evaluation of quasi-static executable slices","year":2023,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Fonds Wetenschappelijk Onderzoek","keywords":"Program slicing; Slicing; Executable; Computer science; Static analysis; Set (abstract data type); Closure (psychology); Programming language; Program analysis; Algorithm","score_opus":0.08142813142564716,"score_gpt":0.3702531606239323,"score_spread":0.2888250291982852,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4322754178","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9904951,0.00034298247,0.0063988552,0.00008096344,0.000025202242,0.00015267481,0.00058751635,0.00041748545,0.0014993594],"genre_scores_gemma":[0.9934284,0.000074821306,0.0051208264,0.000025486814,0.000010240088,0.000073595445,0.00081743253,0.00009847925,0.00035073812],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99015,0.005098532,0.0010682936,0.001061301,0.002304421,0.00031750553],"domain_scores_gemma":[0.6159407,0.31598604,0.0170124,0.023794282,0.024806928,0.0024597605],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011134135,0.000638973,0.0003888217,0.0019935311,0.00072742003,0.0010085929,0.0014634236,0.0010106368,0.0038885728],"category_scores_gemma":[0.17488039,0.0004535174,0.00044088336,0.0013803219,0.0016831348,0.0031526948,0.0012267774,0.000971099,0.0006408982],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.028811187,0.012333826,0.42027554,0.0050260588,0.00095280877,0.0016641346,0.01961497,0.044833638,0.03987033,0.018623073,0.0075476556,0.4004468],"study_design_scores_gemma":[0.0026687298,0.035297293,0.5770494,0.0018168279,0.0019355883,0.0046835523,0.014897265,0.26974326,0.05174451,0.016537273,0.023202928,0.000423405],"about_ca_topic_score_codex":0.00293088,"about_ca_topic_score_gemma":0.004298292,"teacher_disagreement_score":0.011134135,"about_ca_system_score_codex":0.0009767369,"about_ca_system_score_gemma":0.0012189797,"threshold_uncertainty_score":0.058883607},"labels":[],"label_agreement":null},{"id":"W4323316606","doi":"10.1007/s10270-023-01086-5","title":"Efficient regression testing of distributed real-time reactive systems in the context of model-driven development","year":2023,"lang":"en","type":"article","venue":"Software & Systems Modeling","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Regression testing; Context (archaeology); Timestamp; Overhead (engineering); TRACE (psycholinguistics); Distributed computing; Regression; Real-time computing; Programming language; Software development; Statistics","score_opus":0.06037124717234388,"score_gpt":0.2836356329457989,"score_spread":0.22326438577345503,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323316606","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22400095,0.0003290549,0.77040917,0.00039899859,0.000035426223,0.000047805457,0.00005628429,0.002853609,0.0018687058],"genre_scores_gemma":[0.9458537,0.00005033403,0.053468946,0.00002860023,0.000008606261,0.00002011934,0.000057548703,0.00010557747,0.0004065775],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968354,0.0018312614,0.00008931858,0.00034862652,0.00060639286,0.0002890127],"domain_scores_gemma":[0.9888067,0.008507495,0.0006567618,0.0010449666,0.0007961762,0.00018795666],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002127939,0.00062159303,0.0007603728,0.0005414896,0.0003169638,0.0008460318,0.0013664197,0.000558784,0.00094496424],"category_scores_gemma":[0.0138034895,0.00034971468,0.00033926297,0.00041172403,0.00076320395,0.0012718006,0.0010127868,0.00084365305,0.00016112952],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005882535,0.00024224624,0.00457602,0.00019488641,0.00006539866,0.00033306892,0.00019477401,0.8381051,0.020033382,0.017072957,0.0013507444,0.11724323],"study_design_scores_gemma":[0.00001567224,0.000033857013,0.00019069278,0.0000043231544,0.0000069669613,0.00001997057,0.000010538002,0.99337983,0.002910685,0.003307002,0.00011786947,0.0000025346474],"about_ca_topic_score_codex":0.0037447105,"about_ca_topic_score_gemma":0.004531163,"teacher_disagreement_score":0.0037447105,"about_ca_system_score_codex":0.0005878634,"about_ca_system_score_gemma":0.0015595529,"threshold_uncertainty_score":0.011253774},"labels":[],"label_agreement":null},{"id":"W4361986063","doi":"10.1109/tse.2023.3256322","title":"Metamorphic Testing for Web System Security","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Computer science; Oracle; Security testing; Executable; Fuzz testing; Web application; Programming language; Database; Software engineering; Software; World Wide Web; Operating system; Cloud computing; Cloud computing security","score_opus":0.029534783520755,"score_gpt":0.24176754022297112,"score_spread":0.2122327567022161,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4361986063","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022335755,0.00056655006,0.96429014,0.00059374847,0.00004848357,0.00009443628,0.00006618935,0.0025661069,0.009438644],"genre_scores_gemma":[0.56003296,0.0010131899,0.43097922,0.0005649964,0.000119503755,0.00020841158,0.00032620915,0.0005162254,0.0062392764],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973865,0.0008588669,0.00021664938,0.00039384598,0.0009898535,0.0001543055],"domain_scores_gemma":[0.9965299,0.0021349702,0.00032399932,0.0006081317,0.00030687722,0.00009620995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015904034,0.00054745213,0.0004170223,0.0009019571,0.00036719377,0.0011076713,0.00070288574,0.0009227603,0.0028237135],"category_scores_gemma":[0.0056901625,0.00037931692,0.00088035036,0.00050059846,0.0019111462,0.0015981041,0.0014603386,0.0017137448,0.00041194135],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022769965,0.0002040968,0.00417879,0.00040491327,0.00006673832,0.0009551243,0.0005368879,0.11987886,0.043321118,0.5011209,0.0047520744,0.32435277],"study_design_scores_gemma":[0.00004955233,0.00020263049,0.0013414872,0.00024209188,0.000049912695,0.00090314855,0.000060194947,0.69492286,0.028759863,0.2542936,0.019127201,0.000047410627],"about_ca_topic_score_codex":0.0010867806,"about_ca_topic_score_gemma":0.0007908345,"teacher_disagreement_score":0.0028237135,"about_ca_system_score_codex":0.001012533,"about_ca_system_score_gemma":0.00071416487,"threshold_uncertainty_score":0.009446263},"labels":[],"label_agreement":null},{"id":"W4362676437","doi":"10.1145/3586049","title":"Pushing the Limit of 1-Minimality of Language-Agnostic Program Reduction","year":2023,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Debugging; Programming language; Reduction (mathematics); Computer science; Syntax; Implementation; Theoretical computer science; Artificial intelligence; Mathematics","score_opus":0.027250278268288686,"score_gpt":0.3119918759498543,"score_spread":0.2847415976815656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4362676437","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032999896,0.0012279805,0.9450559,0.0024132854,0.00013683658,0.00019623566,0.00013295279,0.007952049,0.009884835],"genre_scores_gemma":[0.26484323,0.00090539304,0.72413737,0.0014630647,0.00015625978,0.0004004718,0.00033479423,0.003849895,0.00390956],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9875048,0.0036379264,0.0008495531,0.0023173927,0.004726996,0.0009633507],"domain_scores_gemma":[0.9657926,0.0178357,0.0021812292,0.0109289475,0.0028183733,0.00044304598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008714531,0.0016349832,0.0012750772,0.0018297848,0.0014423673,0.0034593758,0.0048400573,0.0016694258,0.0032298043],"category_scores_gemma":[0.030450627,0.0015244096,0.0024113194,0.0010404057,0.0065121255,0.008768667,0.00775985,0.0055490355,0.001119099],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005299249,0.00041473983,0.004441574,0.0020986595,0.00021987839,0.0005417875,0.001690357,0.06674919,0.06369172,0.5121967,0.00911694,0.3383086],"study_design_scores_gemma":[0.00015806763,0.0006112027,0.001015372,0.0005570378,0.00028152563,0.0009823708,0.00034976122,0.22924426,0.07417994,0.6370975,0.055343427,0.00017944272],"about_ca_topic_score_codex":0.0014059332,"about_ca_topic_score_gemma":0.0026113235,"teacher_disagreement_score":0.008714531,"about_ca_system_score_codex":0.0019527375,"about_ca_system_score_gemma":0.0040500597,"threshold_uncertainty_score":0.046087384},"labels":[],"label_agreement":null},{"id":"W4367672983","doi":"10.1016/j.jss.2023.111734","title":"GitHub Copilot AI pair programmer: Asset or Liability?","year":2023,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":340,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Polytechnique Montréal","funders":"Fonds de recherche du Québec; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Programmer; Liability; Asset (computer security); Computer science; Computer security; Software engineering; Business; Programming language; Finance","score_opus":0.0401045157013695,"score_gpt":0.30122540886373395,"score_spread":0.2611208931623644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367672983","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007218229,0.0012499846,0.036234368,0.009350669,0.0033549578,0.00038883038,0.010274486,0.09740206,0.8345264],"genre_scores_gemma":[0.028107394,0.0010577029,0.01888845,0.003927327,0.0011281524,0.00021074589,0.008820674,0.027708905,0.9101507],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996239,0.00005316303,0.00001414961,0.000058201087,0.00018199129,0.00006856156],"domain_scores_gemma":[0.99740607,0.0005254421,0.00012656472,0.0007384719,0.0005990751,0.00060448865],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0005709504,0.0009220233,0.00058156124,0.0014697396,0.00075582246,0.0021004216,0.0011473605,0.0013513389,0.64163786],"category_scores_gemma":[0.0047869706,0.00045505297,0.00039723143,0.00094503775,0.00051626883,0.0020690744,0.0027233788,0.0012585464,0.46221107],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009848455,0.000025526864,0.00028403866,0.00007677682,0.0000044664334,0.00021351135,0.000030732837,0.00012492226,0.0006176147,0.0023182882,0.8932247,0.10298088],"study_design_scores_gemma":[0.00006628327,0.00004811429,0.0014120614,0.00018366509,0.000010348483,0.00084939244,0.0000820781,0.0024729848,0.0014472931,0.0076498888,0.985753,0.00002509771],"about_ca_topic_score_codex":0.0020329838,"about_ca_topic_score_gemma":0.004338275,"teacher_disagreement_score":0.64163786,"about_ca_system_score_codex":0.0006686947,"about_ca_system_score_gemma":0.00089231244,"threshold_uncertainty_score":0.5111601},"labels":[],"label_agreement":null},{"id":"W4367692815","doi":"10.48550/arxiv.2305.00083","title":"Reflections on Surrogate-Assisted Search-Based Testing: A Taxonomy and Two Replication Studies based on Industrial ADAS and Simulink Models","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Computer science; Benchmark (surveying); Replication (statistics); Taxonomy (biology); Generalization; Machine learning; Heuristic; Artificial intelligence; Domain (mathematical analysis); Reliability engineering; Engineering","score_opus":0.6868043205396952,"score_gpt":0.3359967428355055,"score_spread":0.3508075777041897,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367692815","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040531594,0.09428142,0.8256965,0.015805503,0.0012496611,0.0014508915,0.00025385743,0.0011416109,0.019588886],"genre_scores_gemma":[0.2947365,0.057121255,0.6370188,0.002961586,0.0008107215,0.0017421519,0.0008257908,0.0005882017,0.0041951016],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9390085,0.027892618,0.0061013955,0.005892171,0.019805443,0.0012998658],"domain_scores_gemma":[0.8291401,0.09396771,0.007007294,0.025600266,0.04254265,0.0017420573],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.049177144,0.0020607067,0.0020941622,0.00875272,0.001878016,0.008310295,0.0070240176,0.004290255,0.0015763553],"category_scores_gemma":[0.12584941,0.0012350584,0.0023161343,0.010442095,0.0074554933,0.016425727,0.00449155,0.00800766,0.00083578244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003251612,0.0006356777,0.008301437,0.0044092913,0.00017593101,0.0004859228,0.009886399,0.03608562,0.007372092,0.21448399,0.010525136,0.70731336],"study_design_scores_gemma":[0.00026557813,0.0050794818,0.0098947575,0.017558048,0.0004493475,0.0043565775,0.017346134,0.2676513,0.034080736,0.25142264,0.39091986,0.0009755675],"about_ca_topic_score_codex":0.008400298,"about_ca_topic_score_gemma":0.003932627,"teacher_disagreement_score":0.95082283,"about_ca_system_score_codex":0.0067369756,"about_ca_system_score_gemma":0.007828368,"threshold_uncertainty_score":0.2600767},"labels":[],"label_agreement":null},{"id":"W4376869521","doi":"10.18280/isi.280211","title":"A Proposed Framework for Identity Verification in Passport Management Using Model Scaling and Semantic Similarity","year":2023,"lang":"en","type":"article","venue":"Ingénierie des systèmes d information","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Similarity (geometry); Computer science; Identity (music); Scaling; Semantic similarity; Natural language processing; Artificial intelligence; Mathematics; Image (mathematics); Physics","score_opus":0.04678460880268017,"score_gpt":0.30477768080344586,"score_spread":0.2579930720007657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376869521","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0054838094,0.00014019037,0.99142516,0.00012886376,0.000026728616,0.00009580826,0.00006270969,0.0010147375,0.0016219304],"genre_scores_gemma":[0.2535414,0.00041377806,0.738878,0.00013561845,0.000071506394,0.0002963366,0.00057959056,0.00017406997,0.005909631],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992719,0.0001395572,0.00004675629,0.00023328747,0.00023704823,0.00007141691],"domain_scores_gemma":[0.99958056,0.000082582206,0.00006599052,0.00009100693,0.0001388745,0.000040995314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009483172,0.00067730754,0.0007468196,0.0010950153,0.0006320408,0.0019008142,0.0019448217,0.0013025574,0.0029225855],"category_scores_gemma":[0.0019173886,0.0003239364,0.0012259135,0.00073349243,0.0006025714,0.0021943685,0.0017339513,0.0011294358,0.0010916606],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017265396,0.00033171,0.004437359,0.00025196176,0.00020643313,0.0006923671,0.0005307044,0.28675455,0.021125194,0.10216904,0.006367979,0.57696],"study_design_scores_gemma":[0.000005220849,0.00006262136,0.00053184445,0.000019530473,0.000024163119,0.00015036274,0.00007999639,0.97430336,0.0043766485,0.01557428,0.00485547,0.00001646864],"about_ca_topic_score_codex":0.007545995,"about_ca_topic_score_gemma":0.006204079,"teacher_disagreement_score":0.007545995,"about_ca_system_score_codex":0.00096800044,"about_ca_system_score_gemma":0.0013725784,"threshold_uncertainty_score":0.015004158},"labels":[],"label_agreement":null},{"id":"W4377231357","doi":"10.1007/978-3-031-33271-5_15","title":"OAMIP: Optimizing ANN Architectures Using Mixed-Integer Programming","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; Université de Montréal","funders":"","keywords":"Computer science; Integer programming; Integer (computer science); Parallel computing; Theoretical computer science; Algorithm; Programming language","score_opus":0.03867590991141418,"score_gpt":0.28250803511291356,"score_spread":0.24383212520149938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4377231357","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010000766,0.0004923336,0.9642087,0.00014482133,0.00018469502,0.00003773929,0.00020651844,0.0066942293,0.018030174],"genre_scores_gemma":[0.13452268,0.0003170173,0.8453905,0.00015787782,0.00006846806,0.0001955838,0.0004940186,0.0022658808,0.016587993],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998299,0.00004110021,0.00000794118,0.000028944483,0.00006955304,0.000022418755],"domain_scores_gemma":[0.9997489,0.00014910041,0.000021097541,0.000027270786,0.000042338605,0.000011361489],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029333844,0.0012819006,0.00045270877,0.00029487896,0.00026341985,0.00081510417,0.0012132152,0.0005810684,0.012896455],"category_scores_gemma":[0.00092920917,0.00058983365,0.0005610759,0.000368437,0.00023289681,0.00075665023,0.0005565417,0.0012035833,0.0022071765],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001697874,0.00007828382,0.00037526383,0.00025765604,0.000090109614,0.00009342151,0.000035523117,0.6967238,0.012082155,0.017321842,0.015601464,0.25717062],"study_design_scores_gemma":[0.000017835724,0.00002158203,0.000053542517,0.00001185408,0.000009463948,0.000019302734,0.000003827575,0.9877965,0.003210354,0.0040671555,0.004783682,0.0000049920372],"about_ca_topic_score_codex":0.0013906792,"about_ca_topic_score_gemma":0.00301422,"teacher_disagreement_score":0.012896455,"about_ca_system_score_codex":0.00028947063,"about_ca_system_score_gemma":0.00042376292,"threshold_uncertainty_score":0.043142915},"labels":[],"label_agreement":null},{"id":"W4378191086","doi":"10.1109/syscon53073.2023.10131048","title":"Enhancing Boofuzz Process Monitoring for Closed-Source SCADA System Fuzzing","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Fuzz testing; SCADA; Computer science; Context (archaeology); Software bug; Process (computing); Embedded system; Software; Computer security; Operating system; Engineering","score_opus":0.029144135486325898,"score_gpt":0.29944532814410946,"score_spread":0.2703011926577836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378191086","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2117609,0.00032841635,0.75155807,0.00022798798,0.000063077954,0.00031922408,0.00016574099,0.03203774,0.0035387946],"genre_scores_gemma":[0.89209694,0.00007689533,0.10590432,0.00008108416,0.000015207854,0.00007844786,0.00020012657,0.00046587636,0.001081068],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9980901,0.00020810314,0.000096710326,0.00039892274,0.0010296602,0.00017644765],"domain_scores_gemma":[0.9956488,0.0013129163,0.0007811564,0.0012602741,0.00083749276,0.0001593938],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014612748,0.0010477379,0.00052779855,0.0014445832,0.00031101942,0.0012412972,0.0015135767,0.00060637586,0.0016689744],"category_scores_gemma":[0.007301755,0.0003666294,0.00047296195,0.00034674365,0.0006623401,0.0019003574,0.001005926,0.0011999026,0.00036988253],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018535465,0.0007990058,0.05236438,0.00045361836,0.00017377728,0.0013101851,0.0016112855,0.13518776,0.28897598,0.014791695,0.0037591974,0.49871957],"study_design_scores_gemma":[0.000053977466,0.00038646747,0.0115030045,0.000047913185,0.00006341246,0.0005020045,0.00009804706,0.8509368,0.12610394,0.003329571,0.0069055683,0.00006928432],"about_ca_topic_score_codex":0.0045715333,"about_ca_topic_score_gemma":0.0033031285,"teacher_disagreement_score":0.0045715333,"about_ca_system_score_codex":0.00091677654,"about_ca_system_score_gemma":0.0010604524,"threshold_uncertainty_score":0.009089828},"labels":[],"label_agreement":null},{"id":"W4378418585","doi":"10.1109/icst57152.2023.00026","title":"Mutation Testing of Deep Reinforcement Learning Based on Real Faults","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Science and Engineering Research Council; Bombardier","keywords":"Reinforcement learning; Mutation; Computer science; Heuristic; Set (abstract data type); Artificial intelligence; Task (project management); Machine learning; Adaptive mutation; Order (exchange); Genetic algorithm; Engineering","score_opus":0.035808777308209985,"score_gpt":0.2859746181044491,"score_spread":0.2501658407962391,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378418585","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63307935,0.00021927596,0.36124322,0.0004200529,0.000058925398,0.000095552095,0.00011786085,0.0019818493,0.0027837853],"genre_scores_gemma":[0.97961456,0.00001773202,0.019818094,0.000057567635,0.0000036164677,0.000028825916,0.000044339715,0.000032532294,0.00038282102],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99844944,0.0005661349,0.00008602033,0.00029036048,0.00035643423,0.00025148308],"domain_scores_gemma":[0.99033105,0.0069561563,0.0007513129,0.00080275803,0.0008590407,0.00029966354],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023700139,0.00059210387,0.00047503857,0.00059228414,0.0002413041,0.0005603937,0.0011634711,0.00077268213,0.001221154],"category_scores_gemma":[0.014579139,0.00021632614,0.00041800234,0.00022684861,0.001408637,0.0010062357,0.00075520226,0.00085216586,0.00010779887],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004000986,0.00019742367,0.009889469,0.00013179149,0.000065405606,0.00036161323,0.00013323443,0.9067916,0.010828268,0.011664371,0.0006639678,0.05887271],"study_design_scores_gemma":[0.000011749599,0.00007464879,0.00032718226,0.000008206736,0.000007302759,0.00002901671,0.00000964435,0.99181575,0.0034669933,0.0041359575,0.000109082546,0.0000044724347],"about_ca_topic_score_codex":0.003163506,"about_ca_topic_score_gemma":0.002192027,"teacher_disagreement_score":0.003163506,"about_ca_system_score_codex":0.0014135736,"about_ca_system_score_gemma":0.001088443,"threshold_uncertainty_score":0.012533963},"labels":[],"label_agreement":null},{"id":"W4378676681","doi":"10.1109/icstw58534.2023.00069","title":"Test Cost Reduction for 5G and Beyond using Machine Learning","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ericsson (Canada); University of Ottawa; Carleton University","funders":"","keywords":"Reduction (mathematics); Test (biology); Computer science; Cost reduction; Machine learning; Artificial intelligence; Mathematics; Geology","score_opus":0.04976053073262066,"score_gpt":0.30632586942066975,"score_spread":0.2565653386880491,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378676681","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36169466,0.0024164694,0.6122404,0.0023393133,0.0001396867,0.00029706987,0.00065658207,0.008355625,0.011860145],"genre_scores_gemma":[0.8392486,0.00032003605,0.15788575,0.00017096226,0.000028048426,0.00007066762,0.00058461475,0.0002391663,0.0014521498],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99839383,0.0005168419,0.0000777529,0.00020411564,0.0006648671,0.00014261232],"domain_scores_gemma":[0.99465454,0.003033714,0.0005018893,0.0007593531,0.0008952265,0.00015530214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014147015,0.0010285343,0.0006627,0.0017054944,0.00028442682,0.0010579828,0.001415533,0.00066035683,0.0037248745],"category_scores_gemma":[0.008126046,0.00018913494,0.00066147966,0.0010362921,0.0003473756,0.0013764572,0.00063787634,0.0008746194,0.00059068284],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005907098,0.00034750885,0.0113078235,0.0002388375,0.00009892476,0.00032603712,0.00006531721,0.23103383,0.02454839,0.0055695665,0.004106984,0.72176605],"study_design_scores_gemma":[0.00004546414,0.00036675634,0.006131324,0.000059867652,0.00006185853,0.00023326094,0.00006208466,0.9632603,0.016710546,0.009861897,0.0031845355,0.00002216303],"about_ca_topic_score_codex":0.005637622,"about_ca_topic_score_gemma":0.005161824,"teacher_disagreement_score":0.005637622,"about_ca_system_score_codex":0.0011922552,"about_ca_system_score_gemma":0.0012872282,"threshold_uncertainty_score":0.012460947},"labels":[],"label_agreement":null},{"id":"W4378676759","doi":"10.1109/icstw58534.2023.00071","title":"On factors that impact the relationship between code coverage and test suite effectiveness: a survey","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Test suite; Test (biology); Suite; Computer science; Code coverage; Empirical research; Code (set theory); Variety (cybernetics); Test case; Programming language; Statistics; Machine learning; Artificial intelligence; Software; Mathematics; Geography","score_opus":0.13644877090685006,"score_gpt":0.35406367320126025,"score_spread":0.2176149022944102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378676759","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40937337,0.55437696,0.016167253,0.003987235,0.00017268087,0.00049013644,0.0030678525,0.00025547316,0.012108984],"genre_scores_gemma":[0.6840571,0.30000743,0.009936317,0.0015313932,0.00019335633,0.0004550394,0.0027532487,0.00016651247,0.0008996447],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9775254,0.0074428353,0.005070553,0.0016525963,0.0077299783,0.0005785029],"domain_scores_gemma":[0.49180132,0.44763806,0.030781927,0.0026353325,0.025602615,0.0015407503],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026397686,0.0005557503,0.0011656649,0.011599692,0.00036731825,0.002166364,0.00082136865,0.0009970972,0.0026477154],"category_scores_gemma":[0.16535845,0.00063142786,0.0020059643,0.013223737,0.0010921515,0.004307751,0.0010199642,0.0011280682,0.00063457526],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034771848,0.00021969024,0.32966712,0.020470448,0.0014547473,0.00024564823,0.0023809802,0.0011976865,0.0013517381,0.0013524368,0.003732037,0.6375798],"study_design_scores_gemma":[0.00006127029,0.0015872744,0.88021654,0.033235483,0.0041486863,0.002771169,0.0065174173,0.0020283726,0.0034889376,0.0025026975,0.06323993,0.00020220615],"about_ca_topic_score_codex":0.002710683,"about_ca_topic_score_gemma":0.0038089189,"teacher_disagreement_score":0.026397686,"about_ca_system_score_codex":0.0014928913,"about_ca_system_score_gemma":0.0023667137,"threshold_uncertainty_score":0.139606},"labels":[],"label_agreement":null},{"id":"W4378676770","doi":"10.1109/icstw58534.2023.00060","title":"Analysis of mutation operators for FSM testing","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Mutation testing; Mutation; Computer science; Operator (biology); Finite-state machine; Software testing; Software fault tolerance; Process (computing); Software; Set (abstract data type); Fault (geology); Mutant; Theoretical computer science; Algorithm; Programming language; Artificial intelligence; Biology; Genetics","score_opus":0.062392869974473346,"score_gpt":0.32285233541539643,"score_spread":0.2604594654409231,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378676770","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5056775,0.0005544146,0.4873362,0.0002527418,0.000041254636,0.00032729702,0.0002913641,0.0022187172,0.0033005173],"genre_scores_gemma":[0.85406893,0.00010390007,0.14450441,0.000062548796,0.000015079063,0.00016561388,0.0003274105,0.00020864658,0.0005434912],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99311924,0.0021911596,0.00033046465,0.00053067494,0.0034855308,0.00034303637],"domain_scores_gemma":[0.95247704,0.040315557,0.0019507832,0.001803469,0.0030319532,0.0004211793],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004839938,0.0008944683,0.0005765493,0.0025929406,0.00050355826,0.00066741527,0.00078754284,0.0009801016,0.0011506802],"category_scores_gemma":[0.035416864,0.00023238153,0.000932497,0.00095414114,0.0012020785,0.0010800904,0.0005864305,0.0009003857,0.000125368],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007742068,0.00053564616,0.022270193,0.00044894827,0.00020751981,0.0007777564,0.0003633784,0.626242,0.09550671,0.03879161,0.001378244,0.2127037],"study_design_scores_gemma":[0.000031175976,0.00021522425,0.0020032516,0.000020763599,0.000041571362,0.00022937807,0.00003392367,0.9681962,0.020057261,0.008546317,0.00060717383,0.000017715593],"about_ca_topic_score_codex":0.0016351412,"about_ca_topic_score_gemma":0.0012309866,"teacher_disagreement_score":0.004839938,"about_ca_system_score_codex":0.0013853468,"about_ca_system_score_gemma":0.0011673016,"threshold_uncertainty_score":0.02559638},"labels":[],"label_agreement":null},{"id":"W4378942602","doi":"10.1145/3597926.3598135","title":"How Effective Are Neural Networks for Fixing Security Vulnerabilities","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":90,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; York University","funders":"National Science Foundation","keywords":"Computer science; Task (project management); Vulnerability (computing); Code (set theory); Software security assurance; Secure coding; Computer security; Source code; Software; Artificial neural network; Software bug; Automation; Programming language; Software engineering; Artificial intelligence; Information security; Engineering; Security service","score_opus":0.038060447194846705,"score_gpt":0.2861846731068167,"score_spread":0.24812422591197003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378942602","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5140172,0.02228146,0.3751123,0.0331829,0.0017553234,0.00016461634,0.001619797,0.0070738257,0.044792604],"genre_scores_gemma":[0.95437783,0.0028822252,0.037387177,0.0009998658,0.00023896592,0.00004452619,0.00047574157,0.0002639316,0.0033298188],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981365,0.00059606956,0.0000737486,0.00053234113,0.00039516267,0.0002661184],"domain_scores_gemma":[0.991687,0.005540934,0.00068671274,0.0009766413,0.0007986865,0.00031004183],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030861732,0.0014066248,0.00074140483,0.0010782357,0.00042263395,0.0023057691,0.00097182835,0.0032346423,0.003046815],"category_scores_gemma":[0.02958437,0.0005349159,0.00043216342,0.0005418243,0.0010822703,0.008299098,0.000807499,0.0020602017,0.0013473346],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010352691,0.00051429955,0.021983,0.00051511737,0.0005800984,0.00013906583,0.00018279806,0.26899907,0.012793669,0.014182462,0.021801667,0.6572735],"study_design_scores_gemma":[0.00008922849,0.00031295826,0.007123369,0.00019598365,0.0002423819,0.00016378966,0.00022554935,0.90876687,0.0148014985,0.062783994,0.0052264603,0.00006792458],"about_ca_topic_score_codex":0.0054576774,"about_ca_topic_score_gemma":0.0067480095,"teacher_disagreement_score":0.0054576774,"about_ca_system_score_codex":0.0009698254,"about_ca_system_score_gemma":0.0011477268,"threshold_uncertainty_score":0.01632148},"labels":[],"label_agreement":null},{"id":"W4379137887","doi":"10.3233/mgs-230023","title":"A testing framework for JADE agent-based software","year":2023,"lang":"en","type":"article","venue":"Multiagent and Grid Systems","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"JADE (particle detector); Computer science; Usability; Multi-agent system; Integration testing; Test strategy; Unit testing; Process (computing); Software; Agent-oriented software engineering; Toolbox; Software engineering; Software development; Artificial intelligence; Human–computer interaction; Operating system; Programming language","score_opus":0.09283915703248641,"score_gpt":0.31041183346596274,"score_spread":0.21757267643347633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379137887","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032682053,0.00018825814,0.9878883,0.00029673506,0.000048639675,0.00023907542,0.0000548778,0.0050036,0.003012337],"genre_scores_gemma":[0.082904875,0.0002840032,0.91244787,0.0001785205,0.000046158646,0.0004790512,0.00025959226,0.0007904831,0.0026093707],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99505484,0.0016942382,0.000686407,0.0005198508,0.0016996944,0.00034484785],"domain_scores_gemma":[0.9932943,0.00282195,0.0005798512,0.0013279682,0.0015885847,0.0003873611],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007088235,0.0009732641,0.0008337819,0.0022861352,0.0009804794,0.0036004558,0.0030409067,0.0017024907,0.0023239253],"category_scores_gemma":[0.01095843,0.00086832384,0.0019045142,0.00068104453,0.002518733,0.0031746016,0.0025067579,0.0033313264,0.0009338359],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015499562,0.00038884726,0.002975696,0.0006642686,0.00011684039,0.0012868309,0.0025848672,0.067905284,0.015731508,0.6894183,0.008946507,0.2098261],"study_design_scores_gemma":[0.00012010705,0.00047007267,0.0020641699,0.0008956281,0.000115131676,0.0027219772,0.0005575691,0.52655315,0.01964388,0.21532229,0.2313247,0.00021141251],"about_ca_topic_score_codex":0.0045090243,"about_ca_topic_score_gemma":0.0035927882,"teacher_disagreement_score":0.007088235,"about_ca_system_score_codex":0.0013970182,"about_ca_system_score_gemma":0.0024687825,"threshold_uncertainty_score":0.037486613},"labels":[],"label_agreement":null},{"id":"W4379537148","doi":"10.1145/3591233","title":"Recursive State Machine Guided Graph Folding for Context-Free Language Reachability","year":2023,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Advanced Research Projects Agency; Defense Advanced Research Projects Agency; National Science Foundation","keywords":"Reachability; Scalability; Computer science; Graph; Theoretical computer science; Path (computing); Mathematics; Algorithm","score_opus":0.028197214482458714,"score_gpt":0.308115621640343,"score_spread":0.2799184071578843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379537148","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025157785,0.00014894665,0.96766764,0.00020161728,0.000036096175,0.00013338626,0.00013452719,0.0037476502,0.0027723873],"genre_scores_gemma":[0.40467843,0.00021906348,0.59059364,0.00021084094,0.00003752226,0.00036842827,0.00068949925,0.0005439695,0.0026585255],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985275,0.00043446044,0.00010493177,0.00048728354,0.00026560994,0.0001800978],"domain_scores_gemma":[0.9970716,0.0017923015,0.00022364469,0.0006330411,0.00022357047,0.000055771696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011451452,0.0007379681,0.00071328063,0.0007615964,0.0007809034,0.0009866063,0.0011643015,0.0011424077,0.004157298],"category_scores_gemma":[0.0058161616,0.0004332142,0.0018742131,0.0007847075,0.002008793,0.002624134,0.0018686712,0.0014342204,0.00094294874],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031203634,0.00019226302,0.0018443015,0.0006523945,0.00007889559,0.0007944256,0.001140957,0.4006468,0.042747382,0.37701488,0.0039813556,0.17059438],"study_design_scores_gemma":[0.000029057264,0.00012347235,0.0002803404,0.000066182954,0.00004781545,0.00015946785,0.000086641485,0.6815243,0.020402858,0.29160482,0.005637262,0.000037819613],"about_ca_topic_score_codex":0.002344118,"about_ca_topic_score_gemma":0.0029652428,"teacher_disagreement_score":0.004157298,"about_ca_system_score_codex":0.0011720334,"about_ca_system_score_gemma":0.0016188872,"threshold_uncertainty_score":0.013907552},"labels":[],"label_agreement":null},{"id":"W4381853525","doi":"10.1016/j.infsof.2023.107286","title":"Reflections on Surrogate-Assisted Search-Based Testing: A Taxonomy and Two Replication Studies based on Industrial ADAS and Simulink Models","year":2023,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; University of Ottawa","funders":"Horizon 2020; HORIZON EUROPE Framework Programme; Natural Sciences and Engineering Research Council of Canada; Horizon 2020 Framework Programme","keywords":"Computer science; Benchmark (surveying); Replication (statistics); Machine learning; Heuristic; Taxonomy (biology); Generalization; Artificial intelligence; Reliability engineering; Search-based software engineering; Software; Engineering; Software system; Programming language","score_opus":0.2884403668754432,"score_gpt":0.3828625042578708,"score_spread":0.09442213738242755,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4381853525","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06672654,0.029288122,0.7137682,0.08880324,0.00088886055,0.00092403166,0.00018584104,0.0006392476,0.09877587],"genre_scores_gemma":[0.7262337,0.017988967,0.23782428,0.0053236187,0.0004556234,0.0009405598,0.0001757751,0.00045884447,0.010598645],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9604964,0.02602384,0.0018826142,0.0017759923,0.0087928185,0.0010282929],"domain_scores_gemma":[0.81545925,0.13573724,0.0035022355,0.027883248,0.016125955,0.0012921055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04377779,0.0010552,0.0016271082,0.0039943983,0.0025638249,0.0093558105,0.0069169933,0.005920365,0.0040963306],"category_scores_gemma":[0.10478088,0.0008424991,0.0012523624,0.0046572248,0.021874981,0.031243483,0.005590797,0.008770133,0.000783935],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001214134,0.00034822547,0.0014486403,0.00037424485,0.00001798464,0.00025796876,0.010162772,0.0073332596,0.00090950617,0.89807177,0.004769397,0.076184794],"study_design_scores_gemma":[0.00012990957,0.00087921583,0.0015536735,0.0017997797,0.00004461619,0.0010079673,0.018325264,0.05870734,0.0057287426,0.7933796,0.11827764,0.00016627765],"about_ca_topic_score_codex":0.012597161,"about_ca_topic_score_gemma":0.005817855,"teacher_disagreement_score":0.04377779,"about_ca_system_score_codex":0.0067421217,"about_ca_system_score_gemma":0.0068682064,"threshold_uncertainty_score":0.23152184},"labels":[],"label_agreement":null},{"id":"W4382930431","doi":"10.1016/j.scico.2023.102990","title":"AmbieGen: A search-based framework for autonomous systems testing","year":2023,"lang":"en","type":"article","venue":"Science of Computer Programming","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Modular design; Software deployment; Autonomous system (mathematics); Architecture; Robot; Scenario testing; Test case; Distributed computing; Artificial intelligence; Software engineering; Machine learning; Programming language","score_opus":0.0676608071989877,"score_gpt":0.3268068828846064,"score_spread":0.2591460756856187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382930431","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007452509,0.0001935575,0.99270105,0.000108129716,0.000019077002,0.0001798499,0.0001336174,0.0044968864,0.0014225907],"genre_scores_gemma":[0.039289586,0.00047815774,0.95619935,0.00015408969,0.000034937155,0.00079017936,0.00083464873,0.00094429066,0.0012747595],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9959014,0.0016221239,0.0003469729,0.00045738177,0.0013797956,0.00029224416],"domain_scores_gemma":[0.99512696,0.003291266,0.00031516192,0.00057553354,0.0005194432,0.00017149476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059237164,0.0024880737,0.0013798586,0.0032420347,0.0008038545,0.002825805,0.005625123,0.002167857,0.00594512],"category_scores_gemma":[0.013446929,0.0013256365,0.0030046413,0.0015159415,0.0029739663,0.0027735876,0.0033388028,0.003704058,0.0016722924],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001877743,0.00026631262,0.0022131489,0.001110172,0.00022998112,0.0006426605,0.00039110435,0.52333283,0.0058501507,0.26095167,0.011219038,0.19360504],"study_design_scores_gemma":[0.00007222302,0.00008445418,0.00020505708,0.00021509134,0.000044473083,0.00029406504,0.00003557273,0.8724615,0.00335928,0.101492584,0.021695375,0.000040324947],"about_ca_topic_score_codex":0.00630772,"about_ca_topic_score_gemma":0.0077366726,"teacher_disagreement_score":0.00630772,"about_ca_system_score_codex":0.0016126035,"about_ca_system_score_gemma":0.0028704992,"threshold_uncertainty_score":0.031327963},"labels":[],"label_agreement":null},{"id":"W4384009688","doi":"10.1109/icse-companion58688.2023.00047","title":"DaMAT: A Data-driven Mutation Analysis Tool","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; European Space Agency","keywords":"Computer science; Interoperability; Mutation testing; Software; Mutation; Source code; Set (abstract data type); Software engineering; Programming language; Operating system","score_opus":0.06459783877168492,"score_gpt":0.32580100139882423,"score_spread":0.26120316262713933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384009688","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009477348,0.00029555528,0.7577058,0.00032681323,0.00017571544,0.00023221286,0.0037463338,0.22364415,0.0043960614],"genre_scores_gemma":[0.17741355,0.0006495887,0.76680434,0.0008688437,0.00009557276,0.001105891,0.0119162155,0.032294333,0.00885158],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99818724,0.00023072823,0.00017498624,0.00031154207,0.0009695039,0.0001258236],"domain_scores_gemma":[0.99595094,0.0023070492,0.00045518286,0.0005224663,0.0006418067,0.00012249444],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018026179,0.0015061472,0.0007201902,0.002693129,0.0006311334,0.0017782433,0.0027234554,0.001364497,0.010476162],"category_scores_gemma":[0.008032192,0.0009126784,0.0014894642,0.0009312237,0.00095929933,0.002200286,0.002065021,0.0019096468,0.003478292],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077295053,0.000551589,0.018935138,0.0023754993,0.00057647587,0.0028032605,0.0010681213,0.0839238,0.08772085,0.047065783,0.17935625,0.57485014],"study_design_scores_gemma":[0.00032881062,0.0002810825,0.0038319165,0.000377124,0.00022644807,0.0025499729,0.0002303568,0.59772754,0.13988073,0.054047573,0.20011999,0.00039849005],"about_ca_topic_score_codex":0.0022153703,"about_ca_topic_score_gemma":0.0027567479,"teacher_disagreement_score":0.010476162,"about_ca_system_score_codex":0.00067430007,"about_ca_system_score_gemma":0.0014617752,"threshold_uncertainty_score":0.03504628},"labels":[],"label_agreement":null},{"id":"W4384026493","doi":"10.1109/icse-companion58688.2023.00087","title":"GLAD: Neural Predicate Synthesis to Repair Omission Faults","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Artificial intelligence; Machine translation; Programming language; Debugger; Predicate (mathematical logic); Generative grammar; Machine learning; Natural language processing; Debugging","score_opus":0.030055209051162008,"score_gpt":0.28562750589366626,"score_spread":0.25557229684250427,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384026493","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016119087,0.00028366238,0.93031275,0.00030739294,0.00010864658,0.00008247978,0.0005596726,0.047225576,0.005000701],"genre_scores_gemma":[0.3159262,0.00024112033,0.66919273,0.00043721733,0.000052033287,0.00018037195,0.0017211022,0.0051333765,0.007115822],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992323,0.00015339915,0.000048470258,0.00020062372,0.00028891576,0.000076359356],"domain_scores_gemma":[0.99835193,0.00086763164,0.00014548891,0.00043545163,0.00016239332,0.000037123824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009572789,0.00090157345,0.00048753354,0.000751281,0.000391607,0.0008560576,0.0017132432,0.0009737038,0.009774969],"category_scores_gemma":[0.003799098,0.0004555273,0.000888457,0.00048170594,0.0011980258,0.0015760199,0.0015647575,0.0015232342,0.0023692313],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032941485,0.0002434479,0.0038572652,0.00076622877,0.000076668744,0.00046873972,0.00035981945,0.2166654,0.045236398,0.045090538,0.03139038,0.6555157],"study_design_scores_gemma":[0.00013406028,0.00023319176,0.0006012544,0.000065806256,0.000060693386,0.00034284525,0.00010726396,0.8648809,0.051559232,0.05093468,0.03103772,0.000042357966],"about_ca_topic_score_codex":0.0021120573,"about_ca_topic_score_gemma":0.004314594,"teacher_disagreement_score":0.009774969,"about_ca_system_score_codex":0.00086364197,"about_ca_system_score_gemma":0.00136537,"threshold_uncertainty_score":0.0327006},"labels":[],"label_agreement":null},{"id":"W4384154545","doi":"10.1145/3597926.3604927","title":"SymRustC: A Hybrid Fuzzer for Rust","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Fuzz testing; Concolic testing; Computer science; Rust (programming language); Benchmark (surveying); Compiler; Programming language; Symbolic execution; Software","score_opus":0.0370533003331452,"score_gpt":0.2912337161585949,"score_spread":0.2541804158254497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384154545","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048342906,0.00036449602,0.87564117,0.00046392583,0.00013469748,0.0002587411,0.00054334255,0.06996409,0.004286709],"genre_scores_gemma":[0.45338708,0.00024000881,0.5326838,0.00078043726,0.00007002335,0.00030788835,0.0014077274,0.0058455397,0.0052773952],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99682224,0.0005253767,0.00025326092,0.00058434316,0.0015893345,0.00022543094],"domain_scores_gemma":[0.9943645,0.0026723307,0.00045851557,0.0014787492,0.0008958023,0.0001301867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023290473,0.0011742667,0.0007090579,0.0019944126,0.0006416112,0.0015739744,0.0020489604,0.0013729236,0.0039340598],"category_scores_gemma":[0.009701776,0.00070195895,0.0013551202,0.0006574036,0.0018963485,0.0029211284,0.002353786,0.0018434796,0.0011014453],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019146218,0.0004329099,0.01897617,0.0011877465,0.00050335034,0.0023341423,0.0016549362,0.13251393,0.1743457,0.08953803,0.03164862,0.5449499],"study_design_scores_gemma":[0.00014416505,0.0004955185,0.00243748,0.00023915114,0.00021132403,0.0015328701,0.0001474852,0.7446831,0.1721122,0.037981022,0.03980039,0.00021523633],"about_ca_topic_score_codex":0.0024456833,"about_ca_topic_score_gemma":0.0027631682,"teacher_disagreement_score":0.0039340598,"about_ca_system_score_codex":0.0011882399,"about_ca_system_score_gemma":0.0013739878,"threshold_uncertainty_score":0.013160706},"labels":[],"label_agreement":null},{"id":"W4384302810","doi":"10.1109/icse48619.2023.00167","title":"Carving UI Tests to Generate API Tests and API Specification","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Application programming interface; Web API; Web application; Web testing; Unit testing; Ajax; Operating system; Test suite; Embedded system; Database; Web server; Web service; Test case; World Wide Web; Software; Web development; The Internet; Web application security; Machine learning","score_opus":0.04563962302134925,"score_gpt":0.2969470074372192,"score_spread":0.2513073844158699,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384302810","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050002944,0.00022473745,0.8803734,0.00024530725,0.0000681245,0.00038424344,0.00050332944,0.06488625,0.0033115584],"genre_scores_gemma":[0.35550758,0.00018386867,0.63091654,0.00057919975,0.000046394587,0.0005690046,0.0026829536,0.006761819,0.0027526107],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99314874,0.0019794137,0.00051438983,0.0011379167,0.0027896075,0.00042994146],"domain_scores_gemma":[0.9608554,0.023974048,0.0022223287,0.0077694124,0.0047867903,0.000391965],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031115543,0.0020132563,0.0011959539,0.0040315934,0.0005660774,0.0021082025,0.0027827134,0.0016811946,0.003618916],"category_scores_gemma":[0.03552447,0.0011098878,0.0014465746,0.0016869652,0.0014659314,0.0024366656,0.0025370354,0.0024331675,0.0021243577],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005759318,0.00055153633,0.022748994,0.0007513064,0.00016642245,0.0016851566,0.0013224639,0.09327107,0.061966307,0.01944702,0.012828503,0.78468525],"study_design_scores_gemma":[0.000093134055,0.00033358886,0.004110345,0.00014711803,0.00010856517,0.0009638699,0.00021799796,0.81489503,0.14403705,0.01617205,0.018790878,0.00013031461],"about_ca_topic_score_codex":0.006480405,"about_ca_topic_score_gemma":0.005095306,"teacher_disagreement_score":0.006480405,"about_ca_system_score_codex":0.0011582667,"about_ca_system_score_gemma":0.0021579529,"threshold_uncertainty_score":0.01645565},"labels":[],"label_agreement":null},{"id":"W4384345629","doi":"10.1109/icse48619.2023.00155","title":"Many-Objective Reinforcement Learning for Online Testing of DNN-Enabled Systems","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Fidelity; Artificial intelligence; High fidelity; Machine learning; Artificial neural network; Human–computer interaction","score_opus":0.05353553344486145,"score_gpt":0.29601844755624895,"score_spread":0.2424829141113875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384345629","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30248728,0.0015615349,0.6856557,0.0007101802,0.00013169201,0.0002839697,0.0002593449,0.004356846,0.0045534614],"genre_scores_gemma":[0.95278966,0.00010235527,0.045893718,0.00017635938,0.000013163945,0.000118124866,0.00018190732,0.0000697085,0.0006549141],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99875116,0.00050823647,0.000091263275,0.00024899092,0.00025357006,0.00014667296],"domain_scores_gemma":[0.99424994,0.0040823007,0.0004890286,0.00029356338,0.0006118036,0.00027335624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028063757,0.0018098567,0.0009676942,0.000558682,0.00032023294,0.00068958197,0.0021883473,0.0013909738,0.0017143665],"category_scores_gemma":[0.0098545775,0.00056225207,0.00052106904,0.0002700733,0.0010461764,0.0014513411,0.0012430033,0.0019935435,0.00024886068],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014311993,0.00014972656,0.0016533629,0.00009539344,0.00003729037,0.00007187386,0.00003023454,0.9649999,0.0011540644,0.00082683674,0.00036563532,0.030472552],"study_design_scores_gemma":[0.000008838352,0.000042146083,0.00008319124,0.0000044823064,0.0000029169705,0.000004929564,0.0000031385866,0.9986274,0.00055582373,0.00060960354,0.000055071334,0.0000023794162],"about_ca_topic_score_codex":0.007898838,"about_ca_topic_score_gemma":0.008502185,"teacher_disagreement_score":0.007898838,"about_ca_system_score_codex":0.0016722295,"about_ca_system_score_gemma":0.0014576473,"threshold_uncertainty_score":0.015705705},"labels":[],"label_agreement":null},{"id":"W4384345653","doi":"10.1109/icse48619.2023.00146","title":"ATM: Black-box Test Case Minimization based on Test Code Similarity and Evolutionary Search","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test suite; Code coverage; Test case; Abstract syntax tree; Fault coverage; Scalability; Minification; Similarity (geometry); Automatic test pattern generation; Test (biology); Java; Regression testing; Syntax; Programming language; Software; Artificial intelligence; Machine learning; Software development; Operating system; Engineering","score_opus":0.04103145356441748,"score_gpt":0.29445215780038997,"score_spread":0.2534207042359725,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384345653","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07816901,0.0009179928,0.9086822,0.00022619047,0.000044636196,0.00057384756,0.0004326726,0.008549668,0.0024038341],"genre_scores_gemma":[0.38362932,0.0002172989,0.61071956,0.00018625686,0.00005069937,0.0007537543,0.002168108,0.00069477875,0.0015802546],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99646986,0.0009767894,0.00024196497,0.00064732373,0.0014468747,0.00021728437],"domain_scores_gemma":[0.99507904,0.0026273427,0.00079803786,0.0007676606,0.0005704617,0.00015736456],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021447213,0.0016459218,0.0014863655,0.0044754827,0.0004883498,0.0010308007,0.0026539504,0.001341997,0.0023567483],"category_scores_gemma":[0.010976999,0.0005655015,0.001801949,0.0025523626,0.0010385761,0.0015530216,0.0015606916,0.000965537,0.00064139743],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000568358,0.00054292195,0.011287791,0.0005395828,0.0004059279,0.00035820363,0.0002060265,0.32633272,0.028146714,0.010051872,0.006240394,0.61531955],"study_design_scores_gemma":[0.00009228667,0.0003063636,0.001959383,0.000032336,0.00007476546,0.00030060852,0.000033927743,0.98232013,0.00776732,0.0051000766,0.001992874,0.000019858657],"about_ca_topic_score_codex":0.0031116134,"about_ca_topic_score_gemma":0.0029707414,"teacher_disagreement_score":0.0044754827,"about_ca_system_score_codex":0.0009725555,"about_ca_system_score_gemma":0.0017109179,"threshold_uncertainty_score":0.0113425255},"labels":[],"label_agreement":null},{"id":"W4384345700","doi":"10.1109/icse48619.2023.00122","title":"GameRTS: A Regression Testing Framework for Video Games","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Regression testing; Game design; Test suite; Sequential game; Video game; Context (archaeology); Game testing; Game Developer; Regression analysis; Software; Machine learning; Artificial intelligence; Test case; Software development; Game theory; Game design document; Programming language; Software construction; Multimedia","score_opus":0.07865055184339448,"score_gpt":0.3418979753851524,"score_spread":0.2632474235417579,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384345700","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0044715106,0.00022425535,0.9430333,0.00018280598,0.00004792057,0.0005511959,0.00042141377,0.05015135,0.0009162607],"genre_scores_gemma":[0.1532527,0.000329116,0.8349888,0.00032557276,0.000086261534,0.0012984194,0.0019375624,0.0060856366,0.0016960071],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98798394,0.005079825,0.001426715,0.0018438773,0.0030198775,0.00064573914],"domain_scores_gemma":[0.980731,0.012606518,0.0019674774,0.0021274171,0.0020139196,0.00055359054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012276102,0.002572714,0.0012949298,0.0045852284,0.00065444084,0.002506213,0.005444078,0.0017987902,0.005180526],"category_scores_gemma":[0.04098045,0.0016179804,0.003545361,0.0012419636,0.0021779444,0.0038283023,0.0033829052,0.0033541229,0.001479306],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011305114,0.0011400667,0.03364904,0.0026635565,0.0010430085,0.0021603573,0.0026698278,0.26558885,0.032762803,0.09787387,0.035383806,0.5239343],"study_design_scores_gemma":[0.00017446859,0.00040718145,0.0026835664,0.0002837352,0.00010918653,0.00075131457,0.00014760304,0.9299191,0.010941261,0.032183025,0.022265365,0.00013420518],"about_ca_topic_score_codex":0.010101323,"about_ca_topic_score_gemma":0.008982886,"teacher_disagreement_score":0.012276102,"about_ca_system_score_codex":0.0014975144,"about_ca_system_score_gemma":0.0030620014,"threshold_uncertainty_score":0.06492299},"labels":[],"label_agreement":null},{"id":"W4384948641","doi":"10.1109/sp46215.2023.10179314","title":"QueryX: Symbolic Query on Decompiled Code for Finding Bugs in COTS Binaries","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Callback; Symbolic execution; Programming language; Scalability; Dead code; Code (set theory); Program analysis; Control flow; Static program analysis; Unreachable code; Symbolic data analysis; Static analysis; Binary code; Theoretical computer science; Source code; Redundant code; Legacy code; Binary number; Code generation; Database; Operating system; Software; Set (abstract data type); Software development","score_opus":0.0654954515918125,"score_gpt":0.3375128509832877,"score_spread":0.2720173993914752,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384948641","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05529028,0.00057366,0.49288368,0.0010222369,0.00015621827,0.00042731842,0.011342671,0.42596194,0.012342045],"genre_scores_gemma":[0.55613565,0.0005394074,0.3477515,0.0009502162,0.00010106315,0.00063202373,0.029996851,0.052621458,0.011271929],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958197,0.00072651275,0.0003483812,0.0006853067,0.002048138,0.00037194896],"domain_scores_gemma":[0.9925298,0.0041426197,0.0006283578,0.0014457296,0.0010560995,0.00019736099],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00243966,0.0026035407,0.0007833668,0.0027709985,0.00076965406,0.0027098686,0.002284711,0.0011169874,0.012893621],"category_scores_gemma":[0.016161855,0.0008465508,0.0012274021,0.0016105681,0.001787006,0.0056256116,0.004864963,0.0012613978,0.0031975051],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037269783,0.00044234676,0.033919957,0.0030652203,0.00029037156,0.0024354123,0.0054902416,0.056199282,0.0804681,0.11001355,0.23052174,0.47342682],"study_design_scores_gemma":[0.00042005986,0.0004439521,0.008142677,0.0003950337,0.00010077567,0.0009226576,0.0012113645,0.68315697,0.11409397,0.07535699,0.11550068,0.00025496722],"about_ca_topic_score_codex":0.00793093,"about_ca_topic_score_gemma":0.008145685,"teacher_disagreement_score":0.012893621,"about_ca_system_score_codex":0.0017355649,"about_ca_system_score_gemma":0.002036835,"threshold_uncertainty_score":0.043133438},"labels":[],"label_agreement":null},{"id":"W4385187421","doi":"10.1109/sp46215.2023.10179420","title":"Examining Zero-Shot Vulnerability Repair with Large Language Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":178,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Zero (linguistics); Computer science; Vulnerability (computing); Ground zero; Computer security; Physics; Linguistics; Nuclear physics","score_opus":0.07209621342546572,"score_gpt":0.3021107687630803,"score_spread":0.23001455533761456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385187421","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89434975,0.0009576162,0.09484602,0.0013848629,0.00011464494,0.00016866457,0.0008431315,0.0049845674,0.0023508933],"genre_scores_gemma":[0.9575008,0.00014002333,0.039097752,0.0002658395,0.000024635341,0.00012117779,0.0014441223,0.00034688596,0.0010588281],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971963,0.0017161071,0.00009787315,0.00053248473,0.00033175488,0.00012550633],"domain_scores_gemma":[0.9561964,0.039122257,0.0010065901,0.0020421802,0.0012852771,0.00034714976],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005121061,0.0010266737,0.0005169607,0.0009871738,0.000526232,0.0009475536,0.0017195106,0.0017143788,0.0015692381],"category_scores_gemma":[0.035737492,0.0005928908,0.00064477295,0.0005223839,0.001446016,0.0024136526,0.0013837717,0.002540188,0.0005392524],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066906837,0.0008243561,0.021550223,0.00067658204,0.00020571359,0.0008983324,0.0021666654,0.84998935,0.008743327,0.004981066,0.008630355,0.10066491],"study_design_scores_gemma":[0.00005899735,0.0002321429,0.0017061229,0.000031713484,0.000026868674,0.00014082315,0.0003684779,0.98742455,0.004125804,0.004283469,0.0015762054,0.000024916428],"about_ca_topic_score_codex":0.008922956,"about_ca_topic_score_gemma":0.014824132,"teacher_disagreement_score":0.008922956,"about_ca_system_score_codex":0.0013516505,"about_ca_system_score_gemma":0.0012209342,"threshold_uncertainty_score":0.027083099},"labels":[],"label_agreement":null},{"id":"W4385301141","doi":"10.1109/serp4iot59158.2023.00013","title":"ra4xstate: An Efficient Quantitative Robustness Analysis Approach for Statecharts","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Robustness (evolution); Computer science; Robustness testing; Timestamp; Distributed computing; Data mining; Real-time computing; Programming language; Software","score_opus":0.0697830445199318,"score_gpt":0.3369201709998566,"score_spread":0.2671371264799248,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385301141","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015660845,0.00005234921,0.99265224,0.000025155108,0.0000074039363,0.000054461772,0.00010193203,0.0052969526,0.00024357613],"genre_scores_gemma":[0.13780639,0.00016721801,0.8583997,0.00006535181,0.00003023153,0.00032115684,0.00067065825,0.0015762108,0.0009630618],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99624395,0.0012591269,0.00026204006,0.0004822459,0.0015158349,0.0002368186],"domain_scores_gemma":[0.99215734,0.0050333044,0.0008308763,0.0009795162,0.000883014,0.000115942494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037602657,0.0018679919,0.001142209,0.00379055,0.00053611235,0.0019010105,0.0022932813,0.0009907514,0.0063051023],"category_scores_gemma":[0.012084424,0.0010739481,0.002993015,0.0009630371,0.0014026306,0.002957931,0.0019261496,0.0018443343,0.00096321275],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037430675,0.00019109061,0.0039569545,0.0012495581,0.00036844518,0.00033784806,0.00039470216,0.5190943,0.045070115,0.097182505,0.004641094,0.3271391],"study_design_scores_gemma":[0.000029595354,0.00012108237,0.0003470595,0.00008320266,0.00005787183,0.000107702945,0.00003265486,0.95121443,0.013564617,0.029321393,0.0050744973,0.00004595001],"about_ca_topic_score_codex":0.0034418374,"about_ca_topic_score_gemma":0.0039663804,"teacher_disagreement_score":0.0063051023,"about_ca_system_score_codex":0.0012915947,"about_ca_system_score_gemma":0.002179656,"threshold_uncertainty_score":0.021092653},"labels":[],"label_agreement":null},{"id":"W4385301350","doi":"10.1109/sbft59156.2023.00011","title":"RIGAA at the SBFT 2023 Tool Competition - Cyber-Physical Systems Track","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Context (archaeology); Track (disk drive); Competition (biology); Reinforcement learning; Cyber-physical system; Artificial intelligence; Operating system; Geology; Ecology","score_opus":0.02585882803850791,"score_gpt":0.27060755010390386,"score_spread":0.24474872206539594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385301350","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1763334,0.005874024,0.42638448,0.008411631,0.013380217,0.0028148121,0.026040172,0.23987281,0.100888535],"genre_scores_gemma":[0.5010759,0.00086181116,0.29458353,0.0019620054,0.0007923322,0.0010643282,0.11485273,0.022938725,0.061868668],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9847739,0.0042905277,0.00060647674,0.0018519292,0.006321091,0.002156084],"domain_scores_gemma":[0.97465986,0.0064849663,0.0006322475,0.003268927,0.009374266,0.0055795913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016830208,0.0030819073,0.0015119282,0.0030457333,0.0014942603,0.004158506,0.0042974493,0.0043319976,0.025282005],"category_scores_gemma":[0.025413355,0.0006747999,0.0021158273,0.0016123826,0.0012778407,0.002899865,0.003669626,0.0036553128,0.015325792],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00263815,0.002527856,0.007990827,0.0013810381,0.000489406,0.001302838,0.0007289653,0.036872707,0.022165377,0.011388601,0.51877743,0.39373684],"study_design_scores_gemma":[0.0022490716,0.0053324243,0.014101225,0.00047325584,0.00020940285,0.0019874428,0.0007098445,0.22848089,0.06486097,0.018184233,0.6630588,0.00035229896],"about_ca_topic_score_codex":0.008875038,"about_ca_topic_score_gemma":0.0094742235,"teacher_disagreement_score":0.025282005,"about_ca_system_score_codex":0.0017035339,"about_ca_system_score_gemma":0.004775685,"threshold_uncertainty_score":0.089007676},"labels":[],"label_agreement":null},{"id":"W4385374119","doi":"10.48550/arxiv.2307.14733","title":"StubCoder: Automated Generation and Repair of Stub Code for Mock Objects","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China; National Science Foundation","keywords":"Stub (electronics); Computer science; Unit testing; Regression testing; Test case; Leverage (statistics); Programming language; Software; Engineering; Structural engineering; Artificial intelligence; Software development; Machine learning; Regression analysis","score_opus":0.1711203268709225,"score_gpt":0.23986957764398073,"score_spread":0.06874925077305824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385374119","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042636015,0.00020472116,0.8683707,0.00012203924,0.00005098095,0.00029089837,0.00039211626,0.08574801,0.0021844713],"genre_scores_gemma":[0.2614598,0.0002006648,0.7200995,0.0001295624,0.000020497871,0.00029814016,0.0024371292,0.010873659,0.004480891],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986533,0.00034593375,0.00008224251,0.00029094756,0.0005064175,0.00012111854],"domain_scores_gemma":[0.9932132,0.0030916135,0.00080169283,0.0018787832,0.00083666237,0.00017807508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017750645,0.0015509237,0.00058178965,0.0016676759,0.00058683835,0.0009063472,0.0020738097,0.0014417822,0.00435913],"category_scores_gemma":[0.008728027,0.00071140355,0.00085259264,0.0005135126,0.0013704246,0.00151706,0.0015242863,0.0009691423,0.0020857342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004362178,0.0005374399,0.015666217,0.00078411895,0.000121416546,0.0016764026,0.0012088,0.076735266,0.14860128,0.016448136,0.026412368,0.7113724],"study_design_scores_gemma":[0.00018808072,0.00056732993,0.0058966912,0.00017492459,0.00008666524,0.002024872,0.00021260128,0.6785174,0.24748869,0.01636245,0.04833327,0.00014700164],"about_ca_topic_score_codex":0.001863134,"about_ca_topic_score_gemma":0.0022895464,"teacher_disagreement_score":0.00435913,"about_ca_system_score_codex":0.0006172907,"about_ca_system_score_gemma":0.0011690218,"threshold_uncertainty_score":0.014582753},"labels":[],"label_agreement":null},{"id":"W4385477690","doi":"10.1109/sp46215.2023.10179438","title":"Finding Specification Blind Spots via Fuzz Testing","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Spec#; Computer science; Codebase; Fuzz testing; Programming language; Code coverage; Source code; Software","score_opus":0.13349067738027193,"score_gpt":0.3164290038649961,"score_spread":0.1829383264847242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385477690","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08405233,0.00029260863,0.906788,0.00077231403,0.00004044302,0.0001476869,0.000110784764,0.005997547,0.0017982614],"genre_scores_gemma":[0.5988767,0.0002386133,0.3979721,0.0004924989,0.000033138967,0.000180364,0.00030849667,0.0007755122,0.0011225826],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98880893,0.0042864936,0.0005193446,0.001509346,0.004216258,0.0006595595],"domain_scores_gemma":[0.9341602,0.0468102,0.0043021324,0.010650764,0.0034968734,0.0005799007],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009166521,0.0011540492,0.0012122962,0.0028367525,0.0008980912,0.0022898612,0.002574583,0.0017600022,0.001854916],"category_scores_gemma":[0.05793146,0.0010971575,0.0016024061,0.0010694833,0.004303513,0.0060953167,0.004214143,0.0025947653,0.0005032998],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010939031,0.00042228115,0.036280412,0.00094901916,0.00041380676,0.0020463967,0.002924416,0.1395347,0.05959134,0.21536751,0.004929677,0.5364466],"study_design_scores_gemma":[0.00012469756,0.00046919522,0.0035687173,0.0002942443,0.00017869359,0.001000587,0.00047553293,0.6851108,0.06521316,0.23537563,0.008061664,0.00012698118],"about_ca_topic_score_codex":0.0026227478,"about_ca_topic_score_gemma":0.0026846637,"teacher_disagreement_score":0.009166521,"about_ca_system_score_codex":0.0014380368,"about_ca_system_score_gemma":0.0025262989,"threshold_uncertainty_score":0.04847777},"labels":[],"label_agreement":null},{"id":"W4385489881","doi":"10.1007/978-3-031-39764-6_14","title":"Granular Traceability Between Requirements and Test Cases for Safety-Critical Software Systems","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Traceability; Computer science; Requirements traceability; Test case; Software engineering; Software; Requirements engineering; Programming language; Requirement","score_opus":0.05911583272759753,"score_gpt":0.3129168050163533,"score_spread":0.2538009722887558,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385489881","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033824556,0.00052869547,0.9532902,0.00026632566,0.00007007982,0.00020467234,0.00017525327,0.0029341634,0.008705964],"genre_scores_gemma":[0.5432642,0.00063080975,0.4470681,0.00016088341,0.000073908086,0.0002597961,0.0011645611,0.00073581515,0.0066420007],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9952539,0.0010735278,0.00036293283,0.00051801937,0.0025205219,0.0002711946],"domain_scores_gemma":[0.9820271,0.010373335,0.0011905865,0.0051482706,0.0009481666,0.00031265023],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004231919,0.00084005727,0.00065463583,0.002824745,0.0005152971,0.0023643337,0.0020530166,0.0010398958,0.004855869],"category_scores_gemma":[0.021004269,0.00086714816,0.0010211329,0.002002688,0.0018148533,0.004678453,0.0023454954,0.002879135,0.000723316],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007543844,0.00058669626,0.0032073553,0.0006924439,0.00013081337,0.00053297763,0.001630868,0.11210674,0.02710811,0.18950483,0.004076037,0.6596688],"study_design_scores_gemma":[0.00013070319,0.0005606005,0.0040163756,0.00043124775,0.000121337835,0.0004146389,0.00028581722,0.3935222,0.019353779,0.5679587,0.013132576,0.000072009745],"about_ca_topic_score_codex":0.0023227646,"about_ca_topic_score_gemma":0.0021228357,"teacher_disagreement_score":0.004855869,"about_ca_system_score_codex":0.0009811999,"about_ca_system_score_gemma":0.001117468,"threshold_uncertainty_score":0.02238077},"labels":[],"label_agreement":null},{"id":"W4385659392","doi":"10.21203/rs.3.rs-3226069/v1","title":"Using Data Mining Techniques to Generate Test Cases from Graph Transformation Systems Specifications","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Test suite; Model-based testing; Change impact analysis; Model transformation; Graph; Regression testing; Automation; Test case; Code coverage; Transformation (genetics); Graph rewriting; Software; Data mining; Software system; Theoretical computer science; Software construction; Programming language; Artificial intelligence; Machine learning; Engineering","score_opus":0.6196641684472338,"score_gpt":0.48093299807095913,"score_spread":0.1387311703762747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385659392","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12055262,0.00025984348,0.866362,0.00032423306,0.000027617534,0.0005797298,0.0021550297,0.007622628,0.0021163186],"genre_scores_gemma":[0.3438938,0.00019967515,0.6483432,0.00007117695,0.0000104179335,0.00050197134,0.006121357,0.00026890571,0.0005896468],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985222,0.00036933113,0.00021426778,0.00024572975,0.0005801585,0.00006836729],"domain_scores_gemma":[0.9904951,0.006485942,0.0007356294,0.0008927245,0.0012912073,0.00009940624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013901342,0.0009868931,0.00051837735,0.004525778,0.00033614365,0.0010938462,0.0012472656,0.0006801646,0.0013732604],"category_scores_gemma":[0.011723313,0.00032839575,0.0009972109,0.0026238956,0.00034130123,0.00093766337,0.0007166417,0.0007405854,0.0005542091],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002785946,0.0007225444,0.027305732,0.00074948877,0.00018005578,0.0008947279,0.00042195505,0.113340974,0.024726463,0.006502403,0.0049075694,0.81996953],"study_design_scores_gemma":[0.000055593187,0.00017021528,0.0034705675,0.00007419961,0.000054509237,0.0004965497,0.00018316864,0.953131,0.032583237,0.006266049,0.0034925425,0.000022395072],"about_ca_topic_score_codex":0.0025460448,"about_ca_topic_score_gemma":0.0043884013,"teacher_disagreement_score":0.004525778,"about_ca_system_score_codex":0.00058337214,"about_ca_system_score_gemma":0.0010746707,"threshold_uncertainty_score":0.0073518157},"labels":[],"label_agreement":null},{"id":"W4385768129","doi":"10.24963/ijcai.2023/542","title":"Revisiting the Evaluation of Deep Learning-Based Compiler Testing","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Compiler; Generator (circuit theory); Correctness; Programming language; Optimizing compiler; Domain-specific language; Compiler construction; Artificial intelligence","score_opus":0.11359556129603447,"score_gpt":0.34106120201102075,"score_spread":0.22746564071498626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385768129","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.65462065,0.015112809,0.20390604,0.003930916,0.00233217,0.0008699616,0.008965599,0.082991734,0.02727024],"genre_scores_gemma":[0.79641473,0.0014144183,0.16981852,0.0020280574,0.00014462812,0.00036180642,0.022013552,0.0026508484,0.0051533994],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9893394,0.0035922078,0.0009228078,0.0023585642,0.0030981137,0.00068895373],"domain_scores_gemma":[0.9824091,0.008622955,0.0009414209,0.0032258814,0.0041575725,0.0006429657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056082862,0.0025579117,0.0011628203,0.0022489694,0.00054788723,0.0018479597,0.0046406873,0.001954155,0.002770092],"category_scores_gemma":[0.032650422,0.00079484633,0.0012236002,0.001654317,0.0020792917,0.0047535757,0.0025525726,0.0033769028,0.001536509],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021829277,0.0013651096,0.023480428,0.0030315074,0.0006659208,0.00043616086,0.00035669675,0.21483965,0.026209908,0.0067714234,0.054622848,0.66603744],"study_design_scores_gemma":[0.00044566946,0.0014269926,0.0077075497,0.00051866175,0.00018569162,0.00040775264,0.00023527457,0.8961121,0.055599205,0.009739805,0.027503064,0.00011823513],"about_ca_topic_score_codex":0.00876802,"about_ca_topic_score_gemma":0.013982462,"teacher_disagreement_score":0.00876802,"about_ca_system_score_codex":0.0029054957,"about_ca_system_score_gemma":0.0034046604,"threshold_uncertainty_score":0.029659808},"labels":[],"label_agreement":null},{"id":"W4385977849","doi":"10.1007/s41870-023-01390-9","title":"A study of the interactive role of metamorphic testing and machine learning in the quality assurance of a deep learning forecasting application","year":2023,"lang":"en","type":"article","venue":"International Journal of Information Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Mitacs","keywords":"Machine learning; Computer science; Artificial intelligence; Metamorphic rock; Generalization; Software quality; Deep learning; Quality (philosophy); Software; Algorithm; Software development; Programming language","score_opus":0.030810263423512226,"score_gpt":0.3040353657264644,"score_spread":0.27322510230295216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385977849","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98044044,0.00011264154,0.016358253,0.00021810138,0.000008997948,0.000019081528,0.000012922504,0.00010412256,0.0027254862],"genre_scores_gemma":[0.99827254,0.000012148787,0.001439489,0.000009055884,0.000002069741,0.0000017429547,0.000005315263,0.0000075109288,0.0002501038],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99933356,0.00027541633,0.000026050131,0.00008979084,0.00019188947,0.000083155246],"domain_scores_gemma":[0.98340005,0.013001222,0.0012632789,0.00081752153,0.0010917821,0.000426179],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013403695,0.00022747932,0.00018829857,0.00043012635,0.00033715012,0.0009142159,0.0006624373,0.0005694526,0.0015732528],"category_scores_gemma":[0.014910487,0.00013899458,0.00015129535,0.00032739353,0.00062045397,0.0011240307,0.00042245653,0.0007624025,0.00008526118],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002405529,0.0029214758,0.18259628,0.0004406991,0.00023066858,0.0033621092,0.004629752,0.31815457,0.13367927,0.06107493,0.0031715892,0.2873331],"study_design_scores_gemma":[0.00002398465,0.0005022683,0.020293202,0.000020100888,0.000031550753,0.00018277379,0.00040344003,0.958448,0.014836946,0.0043844907,0.0008539164,0.0000193037],"about_ca_topic_score_codex":0.0034025963,"about_ca_topic_score_gemma":0.0022795484,"teacher_disagreement_score":0.0034025963,"about_ca_system_score_codex":0.0007005906,"about_ca_system_score_gemma":0.0006052196,"threshold_uncertainty_score":0.007088661},"labels":[],"label_agreement":null},{"id":"W4386031629","doi":"10.1145/3617171","title":"<scp>StubCoder</scp> : Automated Generation and Repair of Stub Code for Mock Objects","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Hong Kong University of Science and Technology; Impact Fund; National Science Foundation","keywords":"Stub (electronics); Computer science; Unit testing; Regression testing; Leverage (statistics); Test case; Programming language; Software; Software development; Artificial intelligence; Engineering; Structural engineering; Machine learning; Software construction","score_opus":0.12252016812216412,"score_gpt":0.3394841656674571,"score_spread":0.21696399754529297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386031629","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015290251,0.00013265044,0.77185965,0.00032820785,0.00007480434,0.00036536134,0.0012941601,0.20523731,0.005417568],"genre_scores_gemma":[0.18242802,0.0003031058,0.7517792,0.0004901771,0.00006828165,0.00048812127,0.008092695,0.04224616,0.0141042555],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99857664,0.00030401628,0.00009884827,0.00028028857,0.00061523094,0.00012510187],"domain_scores_gemma":[0.9919104,0.0028066572,0.0007797426,0.0029720436,0.0013073395,0.00022383375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020528086,0.0016229153,0.0005106226,0.0015195705,0.0008248613,0.0012525164,0.0024315445,0.0016820151,0.011652511],"category_scores_gemma":[0.008709342,0.00089261984,0.0008550024,0.0007322828,0.001895882,0.0023701773,0.001873587,0.0012497546,0.006893845],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005238002,0.00041625247,0.010164193,0.00116461,0.0001384721,0.0023130884,0.001301034,0.035731673,0.12691456,0.025014063,0.16953117,0.626787],"study_design_scores_gemma":[0.0002528722,0.0005934507,0.0071265306,0.0002588009,0.00007812029,0.003621212,0.00017164557,0.43789604,0.32542413,0.020513449,0.20381963,0.00024407925],"about_ca_topic_score_codex":0.0036788129,"about_ca_topic_score_gemma":0.0051545417,"teacher_disagreement_score":0.011652511,"about_ca_system_score_codex":0.0006945487,"about_ca_system_score_gemma":0.0014055255,"threshold_uncertainty_score":0.038981497},"labels":[],"label_agreement":null},{"id":"W4386140507","doi":"10.1007/s10664-023-10363-2","title":"A comparison of reinforcement learning frameworks for software testing tasks","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Regression testing; Software performance testing; Software engineering; System integration testing; Software reliability testing; Software development; Adaptation (eye); Context (archaeology); Test strategy; Reinforcement learning; Software inspection; Manual testing; Software testing; Software construction; Machine learning; Software; Software quality; Programming language","score_opus":0.06973939181903052,"score_gpt":0.35039542706618215,"score_spread":0.28065603524715166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386140507","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2330351,0.003478516,0.7472488,0.00091836974,0.00019646679,0.00047264292,0.00011730628,0.0032867563,0.011246057],"genre_scores_gemma":[0.86274034,0.0008000259,0.13340698,0.00012465256,0.00004897991,0.00025318997,0.00013815227,0.0002589108,0.0022287855],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99500656,0.0027374083,0.0002599138,0.00036493954,0.0012656392,0.00036561972],"domain_scores_gemma":[0.9532519,0.03867261,0.001200451,0.002481481,0.0031976039,0.001195987],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010138228,0.0008326941,0.0012519606,0.0014592473,0.00051089644,0.0015521116,0.0026852398,0.0016819366,0.0024643452],"category_scores_gemma":[0.038140696,0.0004474191,0.00072570733,0.00084340427,0.0010609177,0.0025792255,0.001626477,0.0020853956,0.00049231516],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032479183,0.0019538237,0.0064355596,0.0005297229,0.00020282035,0.000062329906,0.0005686759,0.40932897,0.0026341113,0.042751335,0.0021248013,0.5301599],"study_design_scores_gemma":[0.00020729414,0.0005641889,0.0022070776,0.00005893735,0.000051357805,0.00003704463,0.000079376594,0.9826974,0.0010144345,0.011843474,0.0012069936,0.00003234466],"about_ca_topic_score_codex":0.007389067,"about_ca_topic_score_gemma":0.005608095,"teacher_disagreement_score":0.010138228,"about_ca_system_score_codex":0.002282094,"about_ca_system_score_gemma":0.0028369993,"threshold_uncertainty_score":0.053616703},"labels":[],"label_agreement":null},{"id":"W4386231449","doi":"10.1145/3611643.3616255","title":"Accelerating Continuous Integration with Parallel Batch Testing","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reduction (mathematics); Queue; Regression testing; Variable (mathematics); Test case; Test (biology); Execution time; Software; Parallel computing; Real-time computing; Operating system; Software development; Programming language; Machine learning","score_opus":0.1211693237630664,"score_gpt":0.2987040689830362,"score_spread":0.17753474521996981,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386231449","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21079253,0.0007856673,0.7675107,0.00029385203,0.00013906146,0.0003036414,0.0001310562,0.015609692,0.0044337683],"genre_scores_gemma":[0.63661075,0.00021161183,0.35996363,0.00010597804,0.00004916902,0.00019571379,0.0002917587,0.0006811424,0.0018901954],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99656665,0.00080189586,0.00016096924,0.0006588149,0.001503668,0.0003079575],"domain_scores_gemma":[0.98744595,0.0054905466,0.00094166363,0.0040553072,0.0016906422,0.0003759452],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024959347,0.0012770424,0.00096193794,0.0009716136,0.0004376526,0.0011008165,0.0024651403,0.00070400035,0.0027455336],"category_scores_gemma":[0.01021979,0.0005803337,0.0005677395,0.0010431535,0.00079888984,0.0020000755,0.001417726,0.0015663742,0.000901272],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019210429,0.0012570147,0.013481926,0.0004422363,0.00015485886,0.00043234223,0.00034558115,0.1638462,0.23431332,0.0085374145,0.0049712295,0.5702969],"study_design_scores_gemma":[0.00024675735,0.0016318987,0.0062399614,0.000045444547,0.00011541056,0.00050764816,0.00007497049,0.8003076,0.17523134,0.009198139,0.006320607,0.00008022606],"about_ca_topic_score_codex":0.0025396321,"about_ca_topic_score_gemma":0.0016903944,"teacher_disagreement_score":0.0027455336,"about_ca_system_score_codex":0.00074609584,"about_ca_system_score_gemma":0.0013006155,"threshold_uncertainty_score":0.013199866},"labels":[],"label_agreement":null},{"id":"W4386442953","doi":"10.1145/3617172","title":"On the Caching Schemes to Speed Up Program Reduction","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reduction (mathematics); Debugging; Compiler; Cache; Parallel computing; Process (computing); ENCODE; Memory footprint; Encoding (memory); Compile time; Computation; Theoretical computer science; Computer engineering; Algorithm; Programming language; Artificial intelligence","score_opus":0.1289523923715804,"score_gpt":0.35603718592683653,"score_spread":0.22708479355525613,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386442953","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17682482,0.012198086,0.786121,0.0029232777,0.00027658432,0.0004778983,0.00031968465,0.008951737,0.01190693],"genre_scores_gemma":[0.58480906,0.0037329884,0.40402672,0.0007432545,0.00024121368,0.00040698136,0.00048651957,0.0010158087,0.0045374567],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955525,0.0011731021,0.00033661007,0.0006451344,0.0016188261,0.00067389815],"domain_scores_gemma":[0.97883534,0.009344569,0.0016499566,0.00794138,0.0019233073,0.00030549493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028958225,0.0012965748,0.0010309805,0.0025250558,0.0012544427,0.0019169544,0.0037762655,0.0012105054,0.0034800195],"category_scores_gemma":[0.02042489,0.00081539236,0.0011056093,0.0033947695,0.0028206003,0.00988454,0.002476002,0.0024638718,0.00090893515],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010784277,0.00040533204,0.007951087,0.0010507398,0.0001427265,0.0003756725,0.0009589081,0.10121077,0.059328657,0.25530916,0.013601387,0.5585872],"study_design_scores_gemma":[0.00019734597,0.0009363694,0.0028312942,0.0004631087,0.0003573363,0.0010548909,0.00035812714,0.73592263,0.10947251,0.11066031,0.037577163,0.00016884983],"about_ca_topic_score_codex":0.005671298,"about_ca_topic_score_gemma":0.0061427946,"teacher_disagreement_score":0.005671298,"about_ca_system_score_codex":0.0028819838,"about_ca_system_score_gemma":0.0042944956,"threshold_uncertainty_score":0.020910382},"labels":[],"label_agreement":null},{"id":"W4386515768","doi":"10.1007/978-3-031-40436-8_13","title":"Applying Formal Verification to an Open-Source Real-Time Operating System","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Promela; Computer science; Rotation formalisms in three dimensions; Programming language; Variety (cybernetics); Model checking; Semantics (computer science); Formal methods; Formal specification; Code (set theory); Formal verification; Specification language; Software engineering; Artificial intelligence","score_opus":0.034057987013124694,"score_gpt":0.28208893732041257,"score_spread":0.2480309503072879,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386515768","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019432902,0.0001283809,0.96880764,0.00024452517,0.0001221384,0.000073311814,0.000040656563,0.0048739333,0.0062764804],"genre_scores_gemma":[0.44170937,0.00034153857,0.54698056,0.00019624583,0.00006746402,0.000082641614,0.00017555851,0.0012486029,0.0091980025],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99842834,0.00048284448,0.00010649905,0.00019159666,0.00062539533,0.00016543212],"domain_scores_gemma":[0.9940532,0.0042172056,0.00016978216,0.00081584277,0.0006821245,0.00006190655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020735078,0.0005539308,0.00032671256,0.00046545945,0.00067554857,0.0019008559,0.0012703557,0.00078742654,0.0048008203],"category_scores_gemma":[0.009085616,0.00048404993,0.00098815,0.00025787952,0.0019194924,0.0020858075,0.001421001,0.0015697634,0.0007612232],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034230505,0.00038745915,0.0018892257,0.0008352708,0.00009934882,0.0010118022,0.0015149616,0.13258275,0.070534095,0.2640587,0.0072696796,0.51947427],"study_design_scores_gemma":[0.00013219737,0.00021092348,0.00061681296,0.00020857035,0.00010282755,0.00039348967,0.00021538284,0.6533694,0.10478309,0.20698968,0.03290429,0.00007326851],"about_ca_topic_score_codex":0.005490079,"about_ca_topic_score_gemma":0.0059196856,"teacher_disagreement_score":0.005490079,"about_ca_system_score_codex":0.0009506932,"about_ca_system_score_gemma":0.0020897246,"threshold_uncertainty_score":0.016060352},"labels":[],"label_agreement":null},{"id":"W4386889431","doi":"10.1145/3624745","title":"Search-Based Software Testing Driven by Automatically Generated and Manually Defined Fitness Functions","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada; McMaster University","keywords":"Fitness function; Computer science; Domain (mathematical analysis); Software; Set (abstract data type); Function (biology); Search-based software engineering; Software engineering; Artificial intelligence; Machine learning; Data mining; Software system; Programming language; Software construction; Genetic algorithm","score_opus":0.09860297203966012,"score_gpt":0.31119659442560965,"score_spread":0.21259362238594953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386889431","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1290902,0.00023592387,0.8542691,0.00015449277,0.000030423955,0.00023685559,0.00020836455,0.012783983,0.0029906922],"genre_scores_gemma":[0.6064823,0.00007485714,0.3902706,0.00010249268,0.000011853182,0.00028655076,0.00067996513,0.0009614552,0.0011299254],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958812,0.0017187543,0.00023686494,0.0006239066,0.0012732113,0.00026602452],"domain_scores_gemma":[0.985352,0.010019418,0.0009893216,0.001964724,0.001461238,0.00021332063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035800338,0.0016629044,0.0010458486,0.0023210568,0.00032805404,0.0009935308,0.002078093,0.0012159137,0.0022480728],"category_scores_gemma":[0.018396992,0.0005741015,0.0009089412,0.00087214395,0.0009922803,0.0015472951,0.0013851032,0.0008504344,0.000624601],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066605845,0.0006795098,0.013999213,0.0004405019,0.00020724225,0.00036491704,0.00041563218,0.5785672,0.05101606,0.009334039,0.0035717303,0.3407378],"study_design_scores_gemma":[0.000051508076,0.00011113146,0.00085007946,0.000020209765,0.000017827659,0.00007115717,0.000022623484,0.9882684,0.008176557,0.0017706408,0.0006242563,0.00001555579],"about_ca_topic_score_codex":0.0043889075,"about_ca_topic_score_gemma":0.0056358185,"teacher_disagreement_score":0.0043889075,"about_ca_system_score_codex":0.001024836,"about_ca_system_score_gemma":0.001634885,"threshold_uncertainty_score":0.018933237},"labels":[],"label_agreement":null},{"id":"W4386950100","doi":"10.2139/ssrn.4580607","title":"Hardening of Network Segmentation Using  Automated Referential Penetration Testing","year":2023,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Segmentation; Penetration (warfare); Hardening (computing); Computer science; Penetration test; Artificial intelligence; Reliability engineering; Materials science; Structural engineering; Engineering; Composite material; Operations research","score_opus":0.0736148210369201,"score_gpt":0.3234858426626048,"score_spread":0.24987102162568467,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386950100","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22978525,0.00031380824,0.7521364,0.00023472686,0.00012294135,0.000088407876,0.00009717043,0.012853203,0.0043680845],"genre_scores_gemma":[0.9011264,0.00008058822,0.09620377,0.00008839775,0.000021289923,0.000033619428,0.00011448785,0.0009001329,0.0014314128],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968778,0.0009285255,0.000169144,0.00049405085,0.0011630303,0.0003675054],"domain_scores_gemma":[0.98552203,0.0069514657,0.0011583202,0.0045609395,0.0015650474,0.0002422723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013409915,0.0011113151,0.0009641525,0.0015436037,0.0005427405,0.0012873735,0.0024339894,0.0014608914,0.00503133],"category_scores_gemma":[0.01209255,0.00070278294,0.0007006597,0.00094540865,0.001055669,0.0033337115,0.0019443607,0.0014641281,0.00095329504],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00084942067,0.0004244101,0.006652822,0.00034006828,0.00013707817,0.0015118622,0.0007479729,0.20603517,0.26177108,0.02258169,0.004047944,0.49490055],"study_design_scores_gemma":[0.00003890814,0.00031208238,0.0014450706,0.00003851309,0.000053584637,0.00046811864,0.000084793835,0.8710862,0.11123172,0.012729307,0.0024755625,0.00003612485],"about_ca_topic_score_codex":0.0012300847,"about_ca_topic_score_gemma":0.001492563,"teacher_disagreement_score":0.00503133,"about_ca_system_score_codex":0.0007318405,"about_ca_system_score_gemma":0.0008563665,"threshold_uncertainty_score":0.016831458},"labels":[],"label_agreement":null},{"id":"W4387165217","doi":"10.1007/s10270-023-01123-3","title":"Fault localization in DSLTrans model transformations by combining symbolic execution and spectrum-based analysis","year":2023,"lang":"en","type":"article","venue":"Software & Systems Modeling","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Université de Montréal; Polytechnique Montréal","funders":"Agencia Estatal de Investigación; Austrian Science Fund; Bundesministerium für Digitalisierung und Wirtschaftsstandort; Österreichische Nationalstiftung für Forschung, Technologie und Entwicklung; Ministerio de Ciencia e Innovación; Universidad de Málaga","keywords":"Computer science; Symbolic execution; Programming language; Model checking; Parallel computing; Fault (geology); Theoretical computer science; Software engineering; Software","score_opus":0.022085225537430975,"score_gpt":0.2562257732166029,"score_spread":0.23414054767917192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387165217","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2089312,0.00012470347,0.7753495,0.00018974218,0.00002968681,0.00012714243,0.00022443695,0.01317691,0.0018466937],"genre_scores_gemma":[0.72951925,0.000059804188,0.26869112,0.000041162286,0.000007711355,0.00007082646,0.00041849568,0.0005099416,0.0006817264],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9964586,0.0008084213,0.00025474062,0.0004285367,0.0018368857,0.00021277757],"domain_scores_gemma":[0.9937947,0.002842864,0.00096859963,0.0011090961,0.0011484977,0.00013630847],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016256758,0.0011271815,0.00059657637,0.0025560206,0.00036497955,0.0010310187,0.00092958065,0.0005384984,0.0017631046],"category_scores_gemma":[0.0073448457,0.000275574,0.0009092574,0.0009908463,0.001197108,0.0010867247,0.0013625019,0.00069768494,0.0003423292],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071761897,0.0005276193,0.01817835,0.0004167228,0.00012059573,0.0007264656,0.0005781152,0.5045748,0.12620756,0.018410277,0.0017368783,0.327805],"study_design_scores_gemma":[0.000020999221,0.00009256645,0.0008486396,0.000024325605,0.000018898461,0.000086774555,0.000052999065,0.94855833,0.044518158,0.004727476,0.0010339794,0.000016864453],"about_ca_topic_score_codex":0.0056069144,"about_ca_topic_score_gemma":0.004403157,"teacher_disagreement_score":0.0056069144,"about_ca_system_score_codex":0.0010760131,"about_ca_system_score_gemma":0.0015732222,"threshold_uncertainty_score":0.011148572},"labels":[],"label_agreement":null},{"id":"W4387735187","doi":"10.1145/3628159","title":"Generation-based Differential Fuzzing for Deep Learning Libraries","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; Science, Technology and Innovation Commission of Shenzhen Municipality; National Natural Science Foundation of China","keywords":"Fuzz testing; Computer science; Context (archaeology); Task (project management); Machine learning; Artificial intelligence; Deep learning; Benchmark (surveying); Software engineering; Software; Programming language","score_opus":0.14059899809814053,"score_gpt":0.32340425284178986,"score_spread":0.18280525474364934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387735187","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25136557,0.00085195,0.7299873,0.00079611305,0.000099010096,0.0003395923,0.0004245159,0.012764764,0.003371082],"genre_scores_gemma":[0.840384,0.00012835726,0.15633734,0.00049445545,0.000019891617,0.00020805292,0.00059695804,0.00033870927,0.0014921062],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977336,0.00054848084,0.00018822984,0.0005739673,0.0006290601,0.000326608],"domain_scores_gemma":[0.9938128,0.003779516,0.0005130239,0.0009491626,0.00078264,0.0001629168],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00228105,0.0013306475,0.0008500577,0.0015238872,0.00062123203,0.0010012706,0.002987944,0.0014829491,0.0022390191],"category_scores_gemma":[0.0113723725,0.0006433474,0.0014733388,0.0006277882,0.0018695082,0.0028148955,0.001957968,0.001946809,0.00034414395],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055571424,0.0004470185,0.020447277,0.0004871466,0.00015043005,0.00094709115,0.00042321248,0.50343144,0.031036515,0.019024286,0.0040436764,0.41900608],"study_design_scores_gemma":[0.000035841033,0.000103626124,0.0006717679,0.00002772432,0.000031692525,0.0001152516,0.000028775195,0.9716476,0.013594061,0.012870636,0.00085548573,0.000017436385],"about_ca_topic_score_codex":0.005240703,"about_ca_topic_score_gemma":0.0076090195,"teacher_disagreement_score":0.005240703,"about_ca_system_score_codex":0.0022477435,"about_ca_system_score_gemma":0.0024167842,"threshold_uncertainty_score":0.016308665},"labels":[],"label_agreement":null},{"id":"W4387855793","doi":"10.23977/acss.2023.070813","title":"Research on Vehicle Security Chip Application and Testing Based on Fault Injection","year":2023,"lang":"en","type":"article","venue":"Advances in Computer Signals and Systems","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Fault injection; Automotive industry; Embedded system; Chip; Fault (geology); Computer science; Fault detection and isolation; Automotive engineering; Engineering; Software; Electrical engineering; Operating system; Telecommunications","score_opus":0.06852609827714659,"score_gpt":0.3622269567651739,"score_spread":0.2937008584880273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387855793","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24255447,0.02642743,0.7019857,0.0012356938,0.00041607275,0.00022745373,0.00010783582,0.0015037654,0.025541507],"genre_scores_gemma":[0.9390646,0.010770772,0.04414022,0.0002606983,0.0001407537,0.00008163824,0.00014067696,0.0000807059,0.005319858],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99871564,0.00016761503,0.0000599922,0.00022250027,0.00072148757,0.00011276894],"domain_scores_gemma":[0.99845314,0.0005182327,0.00015384644,0.00014218269,0.0006875961,0.000044956014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00061938807,0.00054977805,0.00047796455,0.0013249088,0.00023364945,0.00069909194,0.0010765576,0.00068937696,0.0013078843],"category_scores_gemma":[0.0017462432,0.00023632187,0.00039172516,0.00088580366,0.00067664107,0.0020532315,0.00034619254,0.00051420735,0.00028999188],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032530611,0.00018925528,0.016020048,0.0016605317,0.00011433136,0.0005936081,0.00068148115,0.04657466,0.3065714,0.05471477,0.0027686562,0.56978595],"study_design_scores_gemma":[0.0000610228,0.0024695378,0.012576345,0.00037952032,0.00028044105,0.003364335,0.0006176818,0.28487194,0.61751145,0.017296478,0.06042487,0.00014640062],"about_ca_topic_score_codex":0.00097059866,"about_ca_topic_score_gemma":0.00049655326,"teacher_disagreement_score":0.0013249088,"about_ca_system_score_codex":0.0006908291,"about_ca_system_score_gemma":0.00064873486,"threshold_uncertainty_score":0.005012393},"labels":[],"label_agreement":null},{"id":"W4388032626","doi":"","title":"Heap Fuzzing:Automatic Garbage Collection Testing with Expert-Guided Random Events","year":2023,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Fuzz testing; Heap (data structure); Computer science; Garbage collection; Random testing; Garbage; Programming language; Database; Machine learning; Test case; Software","score_opus":0.0294070827596225,"score_gpt":0.25298723274485946,"score_spread":0.22358014998523695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388032626","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2172351,0.00017277991,0.74657595,0.00022298076,0.000044256612,0.00037925204,0.00022867344,0.03358471,0.001556336],"genre_scores_gemma":[0.57955927,0.000087336906,0.41556582,0.00018849413,0.0000115561825,0.00025540506,0.00046924799,0.0029076885,0.0009552317],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99480623,0.0019245078,0.00040516132,0.0009637417,0.0015629339,0.00033752384],"domain_scores_gemma":[0.97150964,0.017527467,0.0018571623,0.0071017933,0.0016352601,0.00036859204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044935523,0.0012932498,0.00075023674,0.0010169705,0.0004882828,0.0012456196,0.0033478576,0.0012232303,0.0023191364],"category_scores_gemma":[0.03271895,0.0008376736,0.0007788415,0.00041939935,0.0017563743,0.0028332842,0.0019473481,0.0011715044,0.0006836089],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018283619,0.001297413,0.03631954,0.0012288147,0.00034460117,0.001451758,0.003388819,0.20738615,0.19462368,0.029496925,0.0072572744,0.5153767],"study_design_scores_gemma":[0.0001796768,0.00079880963,0.0037488206,0.00015525565,0.000101150945,0.00063648494,0.00027122744,0.7311231,0.24172819,0.014620458,0.0065127374,0.00012411678],"about_ca_topic_score_codex":0.0011813978,"about_ca_topic_score_gemma":0.0011984851,"teacher_disagreement_score":0.0044935523,"about_ca_system_score_codex":0.00065172336,"about_ca_system_score_gemma":0.0013568049,"threshold_uncertainty_score":0.023764431},"labels":[],"label_agreement":null},{"id":"W4388072373","doi":"10.2316/j.2023.206-0886","title":"PROBABILISTIC MODEL CHECKING METHOD FOR ROBOT PERFORMANCE OPTIMISATION, 461-470.","year":2023,"lang":"en","type":"article","venue":"International Journal of Robotics and Automation","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Probabilistic logic; Model checking; Robot; Statistical model; Artificial intelligence; Programming language","score_opus":0.04118871263207958,"score_gpt":0.32994843460602785,"score_spread":0.2887597219739483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388072373","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002471491,0.00018186329,0.9885306,0.00011925499,0.000050760133,0.000049607737,0.00028173206,0.006635116,0.0016796556],"genre_scores_gemma":[0.27592474,0.00033205524,0.71189237,0.00026159894,0.00008292392,0.00036090755,0.0012089179,0.0024448717,0.0074916743],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996311,0.0011246325,0.00024554713,0.0006095648,0.0014284668,0.00028085746],"domain_scores_gemma":[0.99314284,0.0042595565,0.0002758661,0.0012815252,0.0009506418,0.00008950946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030362252,0.0014737282,0.0013851639,0.0015828513,0.0009778352,0.001551228,0.0030507066,0.001231339,0.009774918],"category_scores_gemma":[0.0116647035,0.0013414238,0.002395476,0.0011010026,0.0017583459,0.0035126882,0.0018460583,0.0026441643,0.0025361087],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088971405,0.00032962824,0.003046841,0.000980588,0.00040860975,0.00039162952,0.00028911917,0.42330176,0.018301925,0.20560792,0.02538065,0.3210716],"study_design_scores_gemma":[0.000094166186,0.00006919621,0.00039686152,0.00006290038,0.000118994314,0.00013974555,0.000023020235,0.9014,0.009941466,0.08055778,0.007159559,0.00003630804],"about_ca_topic_score_codex":0.00725012,"about_ca_topic_score_gemma":0.013621162,"teacher_disagreement_score":0.009774918,"about_ca_system_score_codex":0.0014431484,"about_ca_system_score_gemma":0.0034591965,"threshold_uncertainty_score":0.03270036},"labels":[],"label_agreement":null},{"id":"W4388408996","doi":"10.1109/dsaa60987.2023.10302571","title":"Towards Deep Learning Models for Automatic Computer Program Grading","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Artificial intelligence; Python (programming language); Feature engineering; Leverage (statistics); Deep learning; Machine learning; Implementation; Grading (engineering); Compiler; Programming language; Software engineering","score_opus":0.0649315470941495,"score_gpt":0.313413597167929,"score_spread":0.2484820500737795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388408996","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06133673,0.0010828847,0.92220855,0.00087184337,0.00012699838,0.00013963641,0.0008575126,0.009723338,0.003652569],"genre_scores_gemma":[0.6074382,0.00058475055,0.37729713,0.00070472655,0.000120120865,0.0003661291,0.003996119,0.0005177202,0.008975072],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992914,0.00021339407,0.00004312959,0.00018601524,0.00017423992,0.0000916975],"domain_scores_gemma":[0.99780077,0.0009245302,0.00022934195,0.00026688678,0.00067254464,0.00010598098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014939585,0.0013883509,0.0007665583,0.0012729481,0.00028500392,0.0014167036,0.001973792,0.0013453945,0.002099688],"category_scores_gemma":[0.0066500935,0.00058935705,0.000782263,0.00085579336,0.0005248312,0.0017375447,0.0012175617,0.0029071786,0.001565721],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021576074,0.00033491745,0.0035799698,0.00016209127,0.00010041388,0.00006108547,0.000120278986,0.51286405,0.006502521,0.010192235,0.01297293,0.45289376],"study_design_scores_gemma":[0.0000072477465,0.000022835395,0.00016052318,0.000015297353,0.000006652809,0.0000059280037,0.00000644896,0.99179876,0.0013118262,0.0061004716,0.00055921543,0.0000047824656],"about_ca_topic_score_codex":0.0059634424,"about_ca_topic_score_gemma":0.010905562,"teacher_disagreement_score":0.0059634424,"about_ca_system_score_codex":0.0015047073,"about_ca_system_score_gemma":0.0012692611,"threshold_uncertainty_score":0.01185745},"labels":[],"label_agreement":null},{"id":"W4388483618","doi":"10.1109/ase56229.2023.00148","title":"Towards Autonomous Testing Agents via Conversational Large Language Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Test strategy; Autonomy; Software testing; Process (computing); Software engineering; Process management; Knowledge management; Human–computer interaction; Data science; Software; Engineering; Programming language; Political science","score_opus":0.057643131007728256,"score_gpt":0.2978644873038421,"score_spread":0.24022135629611382,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388483618","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019091448,0.00013477038,0.9720257,0.0013653378,0.000048201975,0.00020781344,0.000028920287,0.0021173807,0.004980436],"genre_scores_gemma":[0.26141557,0.00015038889,0.73320276,0.0006662609,0.00005225082,0.00049902394,0.000118733005,0.00043898405,0.0034560577],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98609644,0.009941474,0.00048516557,0.0009920982,0.0020092616,0.00047550738],"domain_scores_gemma":[0.9727418,0.016589483,0.0016344923,0.004836785,0.0028762936,0.0013212383],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013511097,0.0011083495,0.0006968201,0.0010195723,0.0016030044,0.005021724,0.002995377,0.0032299997,0.0024511814],"category_scores_gemma":[0.04299242,0.0011789926,0.0010464161,0.0004823777,0.004628468,0.008403444,0.007806942,0.004545663,0.001277318],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051223516,0.00059629045,0.0084906705,0.00062267587,0.00013769855,0.0014920043,0.018767796,0.09137108,0.03206412,0.6558124,0.008748897,0.18138424],"study_design_scores_gemma":[0.000111891204,0.00016680695,0.0002902071,0.00014063992,0.000054535423,0.00039344296,0.0015065427,0.7256148,0.011928183,0.22187991,0.03783145,0.00008167366],"about_ca_topic_score_codex":0.0028361604,"about_ca_topic_score_gemma":0.0035059447,"teacher_disagreement_score":0.013511097,"about_ca_system_score_codex":0.0013314639,"about_ca_system_score_gemma":0.0024594187,"threshold_uncertainty_score":0.071454346},"labels":[],"label_agreement":null},{"id":"W4388483694","doi":"10.1109/ase56229.2023.00100","title":"FLUX: Finding Bugs with LLVM IR Based Unit Test Crossovers","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Compiler; Optimizing compiler; Interprocedural optimization; Software bug; Benchmark (surveying); Parallel computing; Fuzz testing; Test suite; Unit testing; Operating system; Programming language; Test case; Software; Loop optimization","score_opus":0.03891028125000218,"score_gpt":0.28167980977821744,"score_spread":0.24276952852821526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388483694","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5505313,0.0014689551,0.34637898,0.00084045116,0.0001991923,0.000501741,0.0019024581,0.09257088,0.0056060576],"genre_scores_gemma":[0.7621904,0.0001900803,0.22919904,0.00050833955,0.00003870328,0.000300956,0.0027054688,0.002790241,0.0020767935],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99680823,0.00073583866,0.00021113858,0.00069050165,0.001274139,0.0002801128],"domain_scores_gemma":[0.9926051,0.0040584225,0.0010814753,0.0012336586,0.0007928667,0.00022841486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030371782,0.0013083634,0.00067953777,0.0024994034,0.000601661,0.00097274233,0.0016769073,0.0012914207,0.0023017654],"category_scores_gemma":[0.016106902,0.00060427294,0.001061755,0.0008239111,0.0013492749,0.0021020484,0.0018115047,0.0009961107,0.000540078],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024548054,0.00083447696,0.16919906,0.0012509333,0.00046383316,0.0028976458,0.002127691,0.09450492,0.13072722,0.016198888,0.030048508,0.54929197],"study_design_scores_gemma":[0.0004500825,0.001970496,0.029059082,0.00030188478,0.0003333326,0.002470348,0.00059292983,0.76976997,0.15606077,0.019472279,0.019312657,0.00020633671],"about_ca_topic_score_codex":0.0028814017,"about_ca_topic_score_gemma":0.0038138074,"teacher_disagreement_score":0.0030371782,"about_ca_system_score_codex":0.0010411314,"about_ca_system_score_gemma":0.0012172834,"threshold_uncertainty_score":0.016062379},"labels":[],"label_agreement":null},{"id":"W4388483831","doi":"10.1109/ase56229.2023.00079","title":"Fuzzing for CPS Mutation Testing","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Entomological Society of America","keywords":"Fuzz testing; Symbolic execution; Computer science; White-box testing; Software testing; Mutation testing; Software reliability testing; Mutation; Non-regression testing; Test strategy; Software performance testing; Software; Test case; Process (computing); Manual testing; Programming language; Software engineering; System integration testing; Reliability engineering; Keyword-driven testing; Software development; Software construction; Machine learning; Engineering","score_opus":0.09621675244336558,"score_gpt":0.3177917592940461,"score_spread":0.2215750068506805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388483831","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13467595,0.0004927139,0.85428894,0.0005173078,0.00008998768,0.00021859053,0.00020783098,0.005757542,0.0037511697],"genre_scores_gemma":[0.72123486,0.00028084,0.27640814,0.00021618552,0.00003218365,0.00014098053,0.00034809476,0.0003692718,0.0009694992],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.996739,0.000837246,0.00017036496,0.0004260146,0.0016195222,0.00020783926],"domain_scores_gemma":[0.99238163,0.0046531195,0.00069998804,0.0012406544,0.00088260655,0.00014192397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021771477,0.0009205118,0.0005810438,0.001600414,0.00058029714,0.0011209617,0.0013188294,0.0009467752,0.002137726],"category_scores_gemma":[0.015117063,0.0003632904,0.0010239639,0.0006532628,0.0017603056,0.0017834608,0.0011964398,0.0013962953,0.00034327072],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004649691,0.0003376389,0.016200896,0.00059025094,0.0001774032,0.0010032008,0.0005555173,0.3918864,0.115054786,0.07957161,0.0034018594,0.39075544],"study_design_scores_gemma":[0.00005256936,0.00023147647,0.0027784964,0.0001342169,0.00009140707,0.00067268894,0.000082449566,0.9102867,0.04715201,0.03224865,0.0062159593,0.000053307307],"about_ca_topic_score_codex":0.0030550081,"about_ca_topic_score_gemma":0.0029068678,"teacher_disagreement_score":0.0030550081,"about_ca_system_score_codex":0.0010408608,"about_ca_system_score_gemma":0.001794224,"threshold_uncertainty_score":0.011514008},"labels":[],"label_agreement":null},{"id":"W4388848588","doi":"10.1145/3631972","title":"Improving Automated Program Repair with Domain Adaptation","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Adaptation (eye); Software engineering; Domain (mathematical analysis)","score_opus":0.07110201268549876,"score_gpt":0.31706774702128526,"score_spread":0.2459657343357865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388848588","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44869503,0.0023910818,0.518744,0.001013421,0.00021714016,0.00035410447,0.00068420573,0.022433562,0.0054674293],"genre_scores_gemma":[0.836714,0.0004856016,0.1574062,0.00055016164,0.000051646955,0.0002282085,0.0017624474,0.00031920723,0.0024825272],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987023,0.00047588395,0.000079459765,0.0004385847,0.00018879803,0.0001149727],"domain_scores_gemma":[0.9961838,0.0019198692,0.000316813,0.00083871715,0.00058959564,0.00015118015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018919746,0.0013656821,0.0007650778,0.0011124415,0.0003322243,0.0007605039,0.0017923223,0.0012954583,0.0012956804],"category_scores_gemma":[0.009198681,0.0004237179,0.000970391,0.0008138012,0.0005095524,0.0022518837,0.0016844687,0.0018891034,0.00075080997],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019496735,0.0006968865,0.010850823,0.00024902695,0.000110477755,0.00015395676,0.00027455974,0.5600482,0.011114928,0.0011909998,0.006024066,0.40909117],"study_design_scores_gemma":[0.000033220946,0.000121159814,0.0012843115,0.0000186991,0.000031494736,0.000055210716,0.00007981269,0.9902293,0.0041747773,0.0017346401,0.0022206015,0.000016707727],"about_ca_topic_score_codex":0.005605346,"about_ca_topic_score_gemma":0.004951067,"teacher_disagreement_score":0.005605346,"about_ca_system_score_codex":0.0008632364,"about_ca_system_score_gemma":0.0014545594,"threshold_uncertainty_score":0.011145413},"labels":[],"label_agreement":null},{"id":"W4388917353","doi":"10.1007/978-981-99-8311-7_15","title":"TorchProbe: Fuzzing Dynamic Deep Learning Compilers","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute; Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Python (programming language); Compiler; Deep learning; Programming language; Artificial intelligence","score_opus":0.019919948858547153,"score_gpt":0.260890147370234,"score_spread":0.24097019851168688,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388917353","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08318569,0.0009333366,0.8202026,0.00067700364,0.00047267904,0.00013453352,0.000570624,0.08369989,0.010123687],"genre_scores_gemma":[0.5985882,0.0003232371,0.37709495,0.00073228724,0.00008331022,0.00014643934,0.0010382467,0.0100002065,0.011993089],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986331,0.00027681186,0.00008828442,0.00029152798,0.000500622,0.00020968616],"domain_scores_gemma":[0.9966074,0.0015631996,0.00019833837,0.0012019721,0.0003567145,0.0000723854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010961012,0.0013933225,0.0006310639,0.0009700735,0.00055112015,0.0010900659,0.002634662,0.0012473803,0.009072074],"category_scores_gemma":[0.00567614,0.0012870186,0.0009388305,0.00068495685,0.0013628172,0.0042091073,0.0025174732,0.0025773947,0.0015740149],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010789289,0.0002449774,0.0037632205,0.00057048915,0.0001915342,0.00038748278,0.000321067,0.17290442,0.052435275,0.051525816,0.030361725,0.68621504],"study_design_scores_gemma":[0.00009919237,0.0002030958,0.00069246977,0.0001534696,0.000096226766,0.0002924854,0.000068923284,0.81646216,0.096318595,0.067483306,0.018066904,0.00006323067],"about_ca_topic_score_codex":0.0022953704,"about_ca_topic_score_gemma":0.0045968955,"teacher_disagreement_score":0.009072074,"about_ca_system_score_codex":0.0013593672,"about_ca_system_score_gemma":0.001577178,"threshold_uncertainty_score":0.030349076},"labels":[],"label_agreement":null},{"id":"W4389158953","doi":"10.1145/3611643.3616272","title":"Statfier: Automated Testing of Static Analyzers via Semantic-Preserving Program Transformations","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Huawei Technologies; National Natural Science Foundation of China","keywords":"Computer science; Static analysis; Heuristics; Key (lock); Test suite; Spectrum analyzer; Spurious relationship; Selection (genetic algorithm); Program analysis; Documentation; Static program analysis; Programming language; Data mining; Test case; Software; Artificial intelligence; Machine learning","score_opus":0.040952538539588405,"score_gpt":0.32100017087266985,"score_spread":0.28004763233308144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389158953","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18998717,0.0003382456,0.6878874,0.00035833853,0.00008737156,0.00043155256,0.0014204214,0.11651372,0.0029758217],"genre_scores_gemma":[0.5549897,0.00014562382,0.43620035,0.00025137447,0.000024263418,0.00027921962,0.00254366,0.0043439,0.0012218321],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99036366,0.0032023715,0.0006825386,0.0014829224,0.0036194858,0.0006490049],"domain_scores_gemma":[0.9782602,0.011888791,0.0022832975,0.0053757196,0.0019314721,0.00026058484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005061737,0.001727064,0.0007926928,0.002147079,0.0006095315,0.001630208,0.0029684461,0.0012571405,0.0020123292],"category_scores_gemma":[0.023514215,0.000986073,0.0013975379,0.0014080254,0.0022446283,0.003035022,0.0014538643,0.0015491061,0.00091563904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020655633,0.0012828585,0.04198906,0.0012477253,0.0005002331,0.0017036552,0.0017075143,0.21199225,0.17937519,0.026389422,0.016269451,0.515477],"study_design_scores_gemma":[0.00024149503,0.00076822424,0.004846613,0.00011125601,0.00012590713,0.00087806565,0.00024319836,0.81953734,0.15177429,0.012226334,0.0091178045,0.00012945617],"about_ca_topic_score_codex":0.00426089,"about_ca_topic_score_gemma":0.0048001097,"teacher_disagreement_score":0.005061737,"about_ca_system_score_codex":0.0012354396,"about_ca_system_score_gemma":0.0032663639,"threshold_uncertainty_score":0.02676934},"labels":[],"label_agreement":null},{"id":"W4389159649","doi":"10.1145/3611643.3616292","title":"Code Coverage Criteria for Asynchronous Programs","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Computer science; Asynchronous communication; JavaScript; Test suite; Asynchrony (computer programming); Software engineering; Metric (unit); Code (set theory); Software quality; Code coverage; Plug-in; Test case; Programming language; Test (biology); Software; Machine learning; Software development","score_opus":0.05903462948133558,"score_gpt":0.3321103564791753,"score_spread":0.2730757269978397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389159649","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47764406,0.0007539836,0.50933796,0.00019889181,0.000038506798,0.00038074245,0.0011291871,0.0028532217,0.007663427],"genre_scores_gemma":[0.91048276,0.00010055927,0.08699616,0.000051557043,0.00004496452,0.00036176192,0.0010869048,0.00024681364,0.00062858895],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9863738,0.0031080574,0.0014710281,0.0013653566,0.006825201,0.0008565583],"domain_scores_gemma":[0.92123616,0.05477523,0.008733557,0.0036135553,0.010093275,0.001548203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053205993,0.0009347697,0.00071061624,0.010206279,0.00056008156,0.001188,0.0011137644,0.0010676521,0.0019085184],"category_scores_gemma":[0.06304105,0.000271398,0.0008092306,0.0021070463,0.0013735639,0.0018291649,0.0018412224,0.00065658276,0.00037158784],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016164114,0.00070101337,0.3156812,0.0011946784,0.00034820012,0.002108027,0.0032241263,0.22712815,0.07979966,0.03900188,0.0048582996,0.3243383],"study_design_scores_gemma":[0.00010421673,0.0012505442,0.10329636,0.00028376636,0.00014262424,0.002039598,0.0006930957,0.80906284,0.04580444,0.030151268,0.0070359013,0.00013530746],"about_ca_topic_score_codex":0.0028266527,"about_ca_topic_score_gemma":0.0023720507,"teacher_disagreement_score":0.010206279,"about_ca_system_score_codex":0.0008641981,"about_ca_system_score_gemma":0.00067960325,"threshold_uncertainty_score":0.02813834},"labels":[],"label_agreement":null},{"id":"W4389161815","doi":"10.1145/3611643.3613872","title":"Prioritizing Natural Language Test Cases Based on Highly-Used Game Features","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Test case; Test (biology); Code coverage; Machine learning; Test suite; Test Management Approach; Natural language; System under test; Artificial intelligence; Data mining; Software; Programming language; Software system","score_opus":0.01845999358637851,"score_gpt":0.28596608037730753,"score_spread":0.267506086790929,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389161815","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5813532,0.0006160451,0.40442467,0.00053601083,0.000090438625,0.00086162117,0.00046972872,0.008067633,0.0035806461],"genre_scores_gemma":[0.8307985,0.00010903738,0.1656104,0.00030847098,0.000021435932,0.0003053209,0.0012336986,0.0003619053,0.0012512892],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9966889,0.0011407064,0.00023787141,0.0006657178,0.0009524675,0.0003143636],"domain_scores_gemma":[0.9846843,0.010971628,0.0012550963,0.00069570815,0.0019987926,0.00039451572],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026866742,0.002109753,0.00096583454,0.0023856359,0.00045547655,0.0013631275,0.0016701913,0.0010744598,0.0015636545],"category_scores_gemma":[0.018778924,0.0005152438,0.0009416766,0.00075074635,0.00084681215,0.00110881,0.00083860446,0.0011865643,0.0004446117],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012039957,0.0018476308,0.055480845,0.0010345432,0.00031714255,0.0026551867,0.0008498651,0.2833122,0.15450658,0.0046440177,0.0065875067,0.48756045],"study_design_scores_gemma":[0.0000906081,0.00044242849,0.008212627,0.00007136894,0.00010304036,0.00050974713,0.00024828737,0.9358744,0.048471067,0.0039990786,0.0019343566,0.00004302031],"about_ca_topic_score_codex":0.007330516,"about_ca_topic_score_gemma":0.013370092,"teacher_disagreement_score":0.007330516,"about_ca_system_score_codex":0.0012845711,"about_ca_system_score_gemma":0.0025637043,"threshold_uncertainty_score":0.01457566},"labels":[],"label_agreement":null},{"id":"W4389162832","doi":"10.1145/3611643.3613101","title":"Ad Hoc Syntax-Guided Program Reduction","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Programming language; Abstract syntax tree; Debugging; Interpreter; Compiler; Grammar; Reduction (mathematics); Syntax; Parsing; Context (archaeology); Natural language processing; Linguistics","score_opus":0.05371215305656096,"score_gpt":0.33135786602655554,"score_spread":0.2776457129699946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389162832","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015735678,0.0002697463,0.9511779,0.00045400488,0.00016664197,0.0004154514,0.00040692286,0.022769827,0.008603794],"genre_scores_gemma":[0.14846776,0.00027766798,0.8268693,0.0008073986,0.00009239209,0.0005021772,0.0016312319,0.008537692,0.012814351],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954268,0.0011460672,0.0002922071,0.0010680835,0.0016622244,0.00040461586],"domain_scores_gemma":[0.98966306,0.0042579407,0.00053086475,0.003644977,0.0017401613,0.00016294203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002581981,0.0015311335,0.0009525234,0.0017592894,0.0010392175,0.0016720138,0.0026012582,0.0010902331,0.009129446],"category_scores_gemma":[0.0132013615,0.0010030459,0.001729118,0.0010549754,0.0024359608,0.0030647307,0.0046011107,0.0032813204,0.0036568083],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00065463904,0.00048142532,0.0036853787,0.0017042342,0.000245117,0.00084861857,0.001923176,0.034636345,0.09444607,0.17883568,0.049708284,0.6328311],"study_design_scores_gemma":[0.0002989803,0.00038830694,0.0015597203,0.00040242137,0.00031291062,0.0013833565,0.00055845216,0.34526646,0.16639438,0.29212734,0.191127,0.00018065167],"about_ca_topic_score_codex":0.001561362,"about_ca_topic_score_gemma":0.002945153,"teacher_disagreement_score":0.009129446,"about_ca_system_score_codex":0.0009867713,"about_ca_system_score_gemma":0.0027507937,"threshold_uncertainty_score":0.030541003},"labels":[],"label_agreement":null},{"id":"W4389165112","doi":"10.1145/3611643.3616275","title":"PPR: Pairwise Program Reduction","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Compiler; Debugging; Benchmark (surveying); Reduction (mathematics); Pairwise comparison; Programming language; Software bug; Parallel computing; Software; Mathematics; Artificial intelligence","score_opus":0.03433554962418357,"score_gpt":0.30685562694519075,"score_spread":0.2725200773210072,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389165112","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047951173,0.0008999204,0.90173584,0.0005374713,0.00018752179,0.0005540081,0.0012462764,0.041187022,0.005700803],"genre_scores_gemma":[0.22784151,0.00039699796,0.7467945,0.0006570841,0.00011505431,0.0009474548,0.0057138517,0.0095392885,0.007994197],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954987,0.0010016931,0.00030870488,0.0010101965,0.0018026056,0.0003780063],"domain_scores_gemma":[0.9943293,0.0016372364,0.0005599797,0.0023277712,0.0010067798,0.00013897437],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020497884,0.002477358,0.00097267923,0.0019288796,0.00081325654,0.00090800726,0.0027981608,0.00089003117,0.0050001014],"category_scores_gemma":[0.008420786,0.00076963386,0.0021349327,0.0015764266,0.0015370314,0.0027111047,0.0034441035,0.0022097686,0.0020505532],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078649767,0.0005090459,0.0092146695,0.001557491,0.0003277925,0.00050191896,0.00062553375,0.04895037,0.07946472,0.031072436,0.048015866,0.7789737],"study_design_scores_gemma":[0.00058390124,0.0024221512,0.008180878,0.00025831387,0.00072310277,0.002609236,0.0005855982,0.52358776,0.1944498,0.12811376,0.13823023,0.0002552815],"about_ca_topic_score_codex":0.0021034312,"about_ca_topic_score_gemma":0.0033479517,"teacher_disagreement_score":0.0050001014,"about_ca_system_score_codex":0.0006921115,"about_ca_system_score_gemma":0.0027887365,"threshold_uncertainty_score":0.01672703},"labels":[],"label_agreement":null},{"id":"W4389205823","doi":"10.22215/etd/2023-15772","title":"Hy2: A Hybrid Vulnerability Analysis Method","year":2023,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Vulnerability (computing); Emulation; Vulnerability assessment; Computer science; Abstraction; Static analysis; Software; Computer security; Programming language","score_opus":0.03027486160120678,"score_gpt":0.36178810521173965,"score_spread":0.33151324361053286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389205823","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0057953773,0.00012888695,0.9693532,0.0002103684,0.000048494163,0.00018007668,0.00053762377,0.020741064,0.0030049526],"genre_scores_gemma":[0.10803564,0.00021550024,0.8748316,0.00030242177,0.000058420737,0.0004193168,0.0015903158,0.005148429,0.009398384],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980958,0.0003330201,0.00012102484,0.00042027366,0.00085680524,0.00017304409],"domain_scores_gemma":[0.9972283,0.0013330422,0.00021287818,0.0006603793,0.0004983226,0.00006700081],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018166942,0.0013907963,0.0007712793,0.0029060678,0.00067269185,0.0014802346,0.0017494872,0.00095620373,0.008247128],"category_scores_gemma":[0.0050496794,0.0010454964,0.0017960302,0.001106235,0.0010255963,0.0037114893,0.0029921292,0.0014366064,0.0020315014],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000387747,0.00019342941,0.011295914,0.0009835034,0.00034965738,0.00075551064,0.0013175447,0.046695996,0.036717597,0.08810011,0.045743022,0.7674599],"study_design_scores_gemma":[0.00021824519,0.00024609652,0.0034964364,0.00027032336,0.00026446633,0.0016107983,0.00038784192,0.6886981,0.059829384,0.110768564,0.13402058,0.00018917669],"about_ca_topic_score_codex":0.0021699157,"about_ca_topic_score_gemma":0.0028320854,"teacher_disagreement_score":0.008247128,"about_ca_system_score_codex":0.0007071603,"about_ca_system_score_gemma":0.002005049,"threshold_uncertainty_score":0.02758944},"labels":[],"label_agreement":null},{"id":"W4389209036","doi":"10.1145/3611643.3616332","title":"A Generative and Mutational Approach for Synthesizing Bug-Exposing Test Cases to Guide Compiler Fuzzing","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"National Natural Science Foundation of China; China Postdoctoral Science Foundation; Tencent","keywords":"Fuzz testing; Computer science; Compiler; Programming language; Compiler correctness; Interprocedural optimization; Optimizing compiler; Code coverage; Compiler construction; Software bug; Toolchain; Test case; Key (lock); Software engineering; Operating system; Loop optimization; Software; Machine learning","score_opus":0.06749669479941017,"score_gpt":0.32009621749096473,"score_spread":0.25259952269155456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389209036","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06518528,0.0005704303,0.9011923,0.0007232683,0.0001000633,0.0005450084,0.0005549614,0.0275616,0.0035670483],"genre_scores_gemma":[0.4004613,0.000272682,0.591182,0.000958314,0.00004970262,0.00057936745,0.0017091599,0.0023211658,0.0024663783],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99620086,0.0013675375,0.00022875094,0.0008316431,0.0011222584,0.00024903248],"domain_scores_gemma":[0.98022443,0.012887232,0.0012376946,0.0038775778,0.0015124502,0.0002605837],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039429623,0.0018954282,0.00074582564,0.002997186,0.00053628895,0.0012282705,0.0030951134,0.0017525585,0.0030942247],"category_scores_gemma":[0.026061969,0.00094328437,0.0016262301,0.0010413998,0.0024536883,0.0020411664,0.002033066,0.0021806913,0.0012642816],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004789769,0.0009986525,0.030996203,0.0010773198,0.00038309602,0.0013241736,0.0012606983,0.30480373,0.07629439,0.028169587,0.014731598,0.5394816],"study_design_scores_gemma":[0.00009687773,0.00034456112,0.002044015,0.00014311445,0.00013068914,0.0005432387,0.000120872755,0.9274818,0.038400054,0.023377823,0.0072552715,0.000061712984],"about_ca_topic_score_codex":0.0028302597,"about_ca_topic_score_gemma":0.0065668775,"teacher_disagreement_score":0.0039429623,"about_ca_system_score_codex":0.0011855452,"about_ca_system_score_gemma":0.0022456495,"threshold_uncertainty_score":0.020852625},"labels":[],"label_agreement":null},{"id":"W4389230915","doi":"10.1109/iavvc57316.2023.10328120","title":"Structured Testing Framework for ADAS Algorithm Development","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; General Motors of Canada; University of Waterloo; U.S. Department of Energy","keywords":"Computer science; Integration testing; Test Management Approach; Test harness; Automation; Protocol (science); Software; Test strategy; Test (biology); White-box testing; Test case; Software engineering; Model-based testing; Development (topology); Software development; Embedded system; Algorithm; Programming language; Software construction; Machine learning; Engineering","score_opus":0.06897894952777034,"score_gpt":0.31434616436116264,"score_spread":0.2453672148333923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389230915","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005078074,0.00004769097,0.9941057,0.00009742012,0.00002073286,0.000099811616,0.000050495793,0.0024838138,0.0025864895],"genre_scores_gemma":[0.03261209,0.0001423369,0.96321833,0.00013819084,0.00003927539,0.0004169947,0.00041954365,0.0006850903,0.0023281032],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9958075,0.0012641533,0.0003878658,0.00043580274,0.001839076,0.00026559827],"domain_scores_gemma":[0.9953277,0.0020735348,0.00028844186,0.00097505236,0.0011046863,0.00023062919],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044301255,0.0010369567,0.0006991516,0.001983506,0.00068127987,0.002177985,0.002865924,0.0013610038,0.00851811],"category_scores_gemma":[0.008457591,0.0007503297,0.0016308818,0.0008410416,0.001824289,0.0017546028,0.002021428,0.0028308118,0.002843107],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011707313,0.00026053394,0.0014780954,0.00049947127,0.00007448721,0.00053639594,0.00057972147,0.09324977,0.009988306,0.572635,0.016101852,0.30447918],"study_design_scores_gemma":[0.00012229772,0.00027257059,0.00072921976,0.00042892335,0.00005695693,0.0008719639,0.00014585478,0.5544896,0.012799041,0.25174353,0.17824544,0.0000946104],"about_ca_topic_score_codex":0.004309136,"about_ca_topic_score_gemma":0.0042119846,"teacher_disagreement_score":0.00851811,"about_ca_system_score_codex":0.0015296603,"about_ca_system_score_gemma":0.0033361,"threshold_uncertainty_score":0.028495908},"labels":[],"label_agreement":null},{"id":"W4389363106","doi":"10.48550/arxiv.2312.00938","title":"WATonoBus: Field-Tested All-Weather Autonomous Shuttle Technology","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Adverse weather; Upstream (networking); Computer science; Modular design; Extreme weather; Architecture; Snow; Enhanced Data Rates for GSM Evolution; Control (management); Perception; Artificial intelligence; Meteorology; Climate change; Geography; Telecommunications","score_opus":0.10394377979326326,"score_gpt":0.21687567386380482,"score_spread":0.11293189407054156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389363106","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70757556,0.0002939365,0.23834321,0.00049420923,0.00032551045,0.00043822563,0.0018747452,0.031964228,0.018690346],"genre_scores_gemma":[0.9518613,0.00005692833,0.042416666,0.00004619188,0.000009394702,0.00010744636,0.0012588879,0.00075555,0.0034876738],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994068,0.00014341365,0.000021725236,0.00007262251,0.00026571253,0.000089776695],"domain_scores_gemma":[0.99883705,0.00029339368,0.00009740951,0.00030427714,0.00032818006,0.00013974294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011206692,0.00065579097,0.00019572637,0.00040846813,0.00028624455,0.00050812797,0.0012451386,0.00033085316,0.00424982],"category_scores_gemma":[0.0022970845,0.00019482891,0.00018112168,0.00025848876,0.0006612408,0.001071498,0.0006403963,0.00075062795,0.00069655385],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027771604,0.0013886049,0.055755366,0.0009908307,0.0002699852,0.0023745752,0.0024090498,0.1781518,0.21976358,0.023839293,0.0585355,0.45374426],"study_design_scores_gemma":[0.0005179861,0.0035747364,0.021035438,0.00011530589,0.00009145427,0.0007148327,0.00071874907,0.7067774,0.19017275,0.007821144,0.06835194,0.000108304885],"about_ca_topic_score_codex":0.0036412696,"about_ca_topic_score_gemma":0.0050334237,"teacher_disagreement_score":0.00424982,"about_ca_system_score_codex":0.0005867503,"about_ca_system_score_gemma":0.00070306764,"threshold_uncertainty_score":0.014217019},"labels":[],"label_agreement":null},{"id":"W4389817810","doi":"10.1007/978-1-4842-9843-5_15","title":"Automated Testing with Unity","year":2023,"lang":"en","type":"book-chapter","venue":"Apress eBooks","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Manitoba Beekeepers' Association","funders":"","keywords":"Computer science","score_opus":0.07365327430997477,"score_gpt":0.25951624547731744,"score_spread":0.18586297116734268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389817810","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019632122,0.0016992647,0.85001105,0.00073765096,0.00037312982,0.00024928516,0.00068026583,0.04856326,0.078053966],"genre_scores_gemma":[0.317182,0.0011651152,0.6316739,0.00059906277,0.00012492371,0.00034589338,0.0024733962,0.007452567,0.038983136],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9943334,0.001619702,0.0002625101,0.0009865619,0.002405761,0.00039204583],"domain_scores_gemma":[0.9931427,0.0037452276,0.0002641,0.0019158857,0.00079307,0.00013884853],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002206216,0.0014171137,0.0009081301,0.0019745522,0.0008140086,0.002542672,0.0030621614,0.0013386115,0.024518045],"category_scores_gemma":[0.0155344205,0.0007486741,0.0010765048,0.0010095225,0.0018178601,0.004326481,0.0042346967,0.0018212836,0.007366052],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028648667,0.00013541018,0.0017415948,0.00059936504,0.000050392788,0.00054056867,0.0011181792,0.017184796,0.020276679,0.13570423,0.04365958,0.7787027],"study_design_scores_gemma":[0.0001489367,0.00036208276,0.0020592732,0.00083656266,0.000084847175,0.0026841082,0.00065130286,0.16714418,0.09862564,0.26823193,0.45899615,0.00017504639],"about_ca_topic_score_codex":0.0018555605,"about_ca_topic_score_gemma":0.0021942612,"teacher_disagreement_score":0.024518045,"about_ca_system_score_codex":0.0011170119,"about_ca_system_score_gemma":0.0010299667,"threshold_uncertainty_score":0.082021},"labels":[],"label_agreement":null},{"id":"W4390091644","doi":"10.48550/arxiv.2312.12604","title":"An empirical study of testing machine learning in the wild","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Oracle; Software reliability testing; White-box testing; Regression testing; Software construction; Software engineering; Test strategy; System integration testing; Software system; Workflow; Software quality assurance; Software quality; Empirical research; Artificial intelligence; Machine learning; Software; Software development; Database; Programming language","score_opus":0.21095865101335456,"score_gpt":0.2677698645024746,"score_spread":0.05681121348912002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390091644","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9882317,0.0003799683,0.007967422,0.000756739,0.000031738145,0.00014164906,0.00065705314,0.00018526762,0.0016484692],"genre_scores_gemma":[0.9907135,0.0001070812,0.0070884405,0.00025977773,0.00003160012,0.00018169792,0.0011678794,0.000100905454,0.00034907646],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.93039334,0.043136142,0.0052296,0.0070566786,0.012526628,0.0016575757],"domain_scores_gemma":[0.3790443,0.52732205,0.034427416,0.032973148,0.021952696,0.004280379],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.038106967,0.0006715626,0.0006916664,0.0036107583,0.0012114235,0.0024664432,0.003027668,0.0020973634,0.0016809321],"category_scores_gemma":[0.26758546,0.000625324,0.0006769971,0.0034396139,0.0055197943,0.0058982717,0.0020630949,0.0030607486,0.0006717239],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008689834,0.0023586247,0.8366539,0.00075889553,0.00034791956,0.0013762047,0.015149698,0.00863536,0.002884276,0.006906654,0.010133886,0.1139256],"study_design_scores_gemma":[0.00041727716,0.004146738,0.71016484,0.0012641626,0.00028460985,0.0049176617,0.026803255,0.18154871,0.012618001,0.022606105,0.03487625,0.00035235673],"about_ca_topic_score_codex":0.0028693161,"about_ca_topic_score_gemma":0.0032888034,"teacher_disagreement_score":0.961893,"about_ca_system_score_codex":0.0017803719,"about_ca_system_score_gemma":0.0011490772,"threshold_uncertainty_score":0.20153129},"labels":[],"label_agreement":null},{"id":"W4390189471","doi":"10.1109/milcom58377.2023.10356308","title":"Input Output Grammar Coverage in Fuzzing","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada; Queen's University","funders":"","keywords":"Fuzz testing; Computer science; Grammar; Code coverage; Code (set theory); Natural language processing; Data mining; Artificial intelligence; Programming language; Set (abstract data type); Software","score_opus":0.032718849871889805,"score_gpt":0.2711694917751658,"score_spread":0.238450641903276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390189471","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7029572,0.00066985574,0.28056002,0.00034217778,0.00004527068,0.0001703784,0.0014759268,0.0062976503,0.0074814656],"genre_scores_gemma":[0.95202345,0.00007880411,0.04627597,0.00004184871,0.000009170981,0.00005333109,0.0008072111,0.00035741838,0.0003528091],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99168164,0.0023510468,0.0005268209,0.0012459575,0.0037284824,0.000466129],"domain_scores_gemma":[0.9356641,0.051187538,0.0029560812,0.0048886035,0.0049248976,0.00037872785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055776294,0.00090905256,0.00078567787,0.0038763946,0.00050699944,0.002212979,0.0010917427,0.0009594145,0.001847966],"category_scores_gemma":[0.06262166,0.00037822247,0.000750398,0.0016641088,0.0015263443,0.0030746784,0.0018112651,0.0009134854,0.00032641366],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017244585,0.0002720896,0.15947329,0.00084479945,0.00028842394,0.0013051784,0.0035140694,0.33073944,0.040771037,0.028161038,0.0036662028,0.4292399],"study_design_scores_gemma":[0.000032771655,0.00035846132,0.028995736,0.00017564211,0.00010710566,0.0007734491,0.00050582486,0.88999736,0.055290323,0.020032508,0.0036487796,0.00008190925],"about_ca_topic_score_codex":0.0053867362,"about_ca_topic_score_gemma":0.0038596182,"teacher_disagreement_score":0.0055776294,"about_ca_system_score_codex":0.0012121836,"about_ca_system_score_gemma":0.0010389853,"threshold_uncertainty_score":0.029497623},"labels":[],"label_agreement":null},{"id":"W4390679819","doi":"10.1109/dsc59305.2023.00078","title":"FixGPT: A Novel Three-Tier Deep Learning Model for Automated Program Repair","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University","funders":"","keywords":"Computer science; Identifier; Artificial intelligence; Machine learning; Deep learning; Benchmarking; Transformer; Data mining; Programming language","score_opus":0.053623571966626885,"score_gpt":0.3244303272742805,"score_spread":0.27080675530765363,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390679819","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061266124,0.001232954,0.92082155,0.0007611882,0.00016305978,0.00015209906,0.000968196,0.011869348,0.0027654255],"genre_scores_gemma":[0.685345,0.0007760285,0.29707137,0.0010762325,0.000100618665,0.000366572,0.004155424,0.00052996556,0.010578762],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997708,0.00003489454,0.000012147891,0.000084156214,0.000053000582,0.000044888624],"domain_scores_gemma":[0.9996394,0.0001309677,0.000032078067,0.00005995453,0.00010475769,0.00003287296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006029785,0.0012046908,0.00068818755,0.0006696478,0.00027352446,0.0008241357,0.0027602292,0.0014381785,0.0024873693],"category_scores_gemma":[0.0015837861,0.0005056573,0.0008422919,0.00058326987,0.0005415693,0.0018077365,0.0012522927,0.0021007664,0.00077796547],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021326491,0.00026216428,0.0046355315,0.00014477175,0.0001575955,0.0001918774,0.00009157875,0.62601143,0.0057095275,0.005250173,0.013080907,0.34425122],"study_design_scores_gemma":[0.0000072790044,0.000030583244,0.00011135179,0.000005659242,0.0000097915445,0.000020533855,0.0000037549808,0.9970913,0.0006833114,0.0015345839,0.00049764465,0.0000041509184],"about_ca_topic_score_codex":0.008507031,"about_ca_topic_score_gemma":0.0129157845,"teacher_disagreement_score":0.008507031,"about_ca_system_score_codex":0.0010800719,"about_ca_system_score_gemma":0.0014101926,"threshold_uncertainty_score":0.016915083},"labels":[],"label_agreement":null},{"id":"W4390824862","doi":"10.23977/jeis.2023.080615","title":"Research on the requirement decomposition and test verification for vehicle data security","year":2023,"lang":"en","type":"article","venue":"Journal of Electronics and Information Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Process (computing); Automotive industry; Test data; Decomposition; Data mining; Software engineering; Engineering","score_opus":0.11930741518378968,"score_gpt":0.41757559014321943,"score_spread":0.29826817495942975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390824862","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033643514,0.00053271,0.96068454,0.0005982437,0.000029170465,0.0003093742,0.00006876113,0.00055057195,0.0035832499],"genre_scores_gemma":[0.36398625,0.0009298492,0.63170266,0.00026884125,0.000036454716,0.00030600542,0.00044735824,0.00023050925,0.0020920702],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9810301,0.0078022946,0.0014705706,0.0018847949,0.007032105,0.00078007585],"domain_scores_gemma":[0.93862,0.03757189,0.0044084876,0.00904355,0.009732737,0.0006234294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010526489,0.0010033031,0.0006884135,0.002604332,0.00065230904,0.002403833,0.0019045061,0.00092926325,0.0023779357],"category_scores_gemma":[0.04924693,0.0007321481,0.0015026976,0.00157623,0.0020916564,0.0056469026,0.0013280774,0.002434102,0.0004395619],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029313596,0.00061339076,0.012504576,0.0015636005,0.00016302789,0.0007100396,0.0025748874,0.06755771,0.06733767,0.15970327,0.0025902104,0.68438846],"study_design_scores_gemma":[0.0000721214,0.0009798162,0.008610019,0.00095973007,0.000213498,0.0019673714,0.0023256345,0.71493524,0.118328676,0.12309588,0.028373107,0.00013897165],"about_ca_topic_score_codex":0.004449024,"about_ca_topic_score_gemma":0.0032190813,"teacher_disagreement_score":0.010526489,"about_ca_system_score_codex":0.002011706,"about_ca_system_score_gemma":0.0041240975,"threshold_uncertainty_score":0.055670083},"labels":[],"label_agreement":null},{"id":"W4391211964","doi":"10.48550/arxiv.2401.12364","title":"Guiding the Search Towards Failure-Inducing Test Inputs Using Support Vector Machines","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Support vector machine; Sorting; Computer science; Machine learning; Artificial intelligence; Structured support vector machine; Genetic algorithm; Data mining; Pattern recognition (psychology); Algorithm","score_opus":0.15501503968045655,"score_gpt":0.24686351685003938,"score_spread":0.09184847716958283,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391211964","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12568931,0.0005030958,0.8695516,0.00050711876,0.00004722523,0.0001065477,0.00007041345,0.0016525234,0.0018721911],"genre_scores_gemma":[0.8236136,0.00011134119,0.1745324,0.0002460052,0.000022516595,0.00013882587,0.00017923201,0.00013165486,0.0010244425],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99830854,0.0007592054,0.00007846101,0.00027598767,0.00042325436,0.0001545243],"domain_scores_gemma":[0.9933211,0.0051106843,0.0004825381,0.00029797966,0.00061632396,0.00017130049],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019410029,0.0014381252,0.0010483215,0.0012033553,0.00037944337,0.00083324016,0.0017185238,0.0013283645,0.0013776467],"category_scores_gemma":[0.0111092925,0.00047032378,0.000744385,0.00044016988,0.0009996855,0.0010328646,0.0010468957,0.0014378066,0.0003696818],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001293802,0.00017604552,0.004771312,0.00011097959,0.00006341756,0.00020400147,0.000109865134,0.86138153,0.0046153804,0.004616532,0.0010504226,0.12277114],"study_design_scores_gemma":[0.000014706055,0.00008653357,0.00017332505,0.000012698683,0.000009141182,0.000033529115,0.00001827453,0.99536777,0.001432553,0.0025947448,0.00025174697,0.0000050814474],"about_ca_topic_score_codex":0.0028487106,"about_ca_topic_score_gemma":0.0028435492,"teacher_disagreement_score":0.0028487106,"about_ca_system_score_codex":0.0007172508,"about_ca_system_score_gemma":0.0014806498,"threshold_uncertainty_score":0.0102651715},"labels":[],"label_agreement":null},{"id":"W4391462747","doi":"10.1109/ms.2024.3418570","title":"Generative AI to Generate Test Data Generators","year":2024,"lang":"en","type":"preprint","venue":"IEEE Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Generative grammar; Test (biology); Computer science; Artificial intelligence; Test data; Natural language processing; Programming language","score_opus":0.06607790163968237,"score_gpt":0.3343913292862057,"score_spread":0.26831342764652333,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391462747","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040233554,0.000042900603,0.992638,0.0001644012,0.000022577791,0.00008417661,0.0000814334,0.0010534462,0.0018896936],"genre_scores_gemma":[0.21401827,0.00012697137,0.77928007,0.00038891102,0.000058551293,0.0007108093,0.0007246843,0.0009261003,0.003765614],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961234,0.0020469602,0.0001770249,0.0005155173,0.00096532656,0.00017184009],"domain_scores_gemma":[0.9796188,0.014717371,0.00041488305,0.003780907,0.0012582595,0.00020965862],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045090797,0.00094772736,0.00070638204,0.0019208073,0.0005450813,0.0015882818,0.0024188014,0.0011168146,0.004005122],"category_scores_gemma":[0.024569819,0.00064203626,0.0014366906,0.0013204487,0.0026151773,0.0018878127,0.002183272,0.0024749378,0.0011508683],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019302045,0.00026175453,0.0040332098,0.00042417992,0.0001690585,0.00041713452,0.0014924059,0.2378835,0.017692361,0.45954832,0.007741562,0.2701435],"study_design_scores_gemma":[0.00005947733,0.00007572523,0.00029398396,0.00004293595,0.000033073477,0.00019646483,0.000077054356,0.7836094,0.009780834,0.19909996,0.006700213,0.00003088275],"about_ca_topic_score_codex":0.0012590212,"about_ca_topic_score_gemma":0.0023844368,"teacher_disagreement_score":0.0045090797,"about_ca_system_score_codex":0.0010333463,"about_ca_system_score_gemma":0.0010127992,"threshold_uncertainty_score":0.023846567},"labels":[],"label_agreement":null},{"id":"W4391614764","doi":"10.1145/3644388","title":"DeepGD: A Multi-Objective Black-Box Test Selection Approach for Deep Neural Networks","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Selection (genetic algorithm); Black box; Artificial neural network; Artificial intelligence; Deep neural networks; Machine learning; Test (biology)","score_opus":0.0672593941211773,"score_gpt":0.31481957568626506,"score_spread":0.24756018156508774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391614764","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049873948,0.00068242097,0.94371265,0.0004627037,0.000051144514,0.00023231767,0.00023119648,0.0034657202,0.0012878875],"genre_scores_gemma":[0.64526355,0.00018682358,0.34937075,0.0007682182,0.0000613251,0.0005431295,0.0010149236,0.00040206118,0.0023892242],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99776804,0.00093854335,0.00011121885,0.00041667675,0.00052277796,0.0002427777],"domain_scores_gemma":[0.993494,0.004433055,0.0004658075,0.00035852398,0.00096739276,0.00028128232],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036264842,0.0024101313,0.0018573055,0.00183162,0.00044283015,0.00083255686,0.0030960464,0.0016944687,0.0027045864],"category_scores_gemma":[0.00862855,0.00096080184,0.0010263422,0.00073695404,0.0013239811,0.0015369436,0.0021699932,0.002335218,0.00036490892],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000260118,0.00022038464,0.0037563501,0.00015573484,0.00013656185,0.00019892516,0.00006972935,0.8263527,0.0041572843,0.0033590177,0.002566647,0.1587666],"study_design_scores_gemma":[0.000022434702,0.00006834147,0.0001812957,0.000007921684,0.000009743893,0.000019740393,0.000007552403,0.9961738,0.0012414162,0.00206919,0.00019304828,0.000005542646],"about_ca_topic_score_codex":0.0065109967,"about_ca_topic_score_gemma":0.0091484785,"teacher_disagreement_score":0.0065109967,"about_ca_system_score_codex":0.0020721043,"about_ca_system_score_gemma":0.0026796015,"threshold_uncertainty_score":0.019178867},"labels":[],"label_agreement":null},{"id":"W4391941017","doi":"10.1016/j.jnca.2024.103851","title":"Hardening of network segmentation using automated referential penetration testing","year":2024,"lang":"en","type":"article","venue":"Journal of Network and Computer Applications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Segmentation; Penetration (warfare); Hardening (computing); Artificial intelligence; Computer vision; Composite material; Materials science; Operations research","score_opus":0.037988586729007415,"score_gpt":0.3028060182771774,"score_spread":0.26481743154817,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391941017","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16278599,0.00037640258,0.82997465,0.00020207358,0.000037634818,0.00013762241,0.000057033896,0.0026386788,0.0037899169],"genre_scores_gemma":[0.86363494,0.0002444242,0.13489616,0.00007827182,0.000018074528,0.000081191116,0.00010442569,0.00021510047,0.0007273453],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9958047,0.0014886541,0.00014862341,0.0005009415,0.0016339006,0.00042322054],"domain_scores_gemma":[0.9894002,0.006158582,0.0016602209,0.0014649888,0.0011436088,0.00017245226],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027002709,0.001327968,0.00088294956,0.0024090712,0.00047315384,0.0014488195,0.0016832462,0.0012643472,0.001311152],"category_scores_gemma":[0.011221163,0.00051763176,0.0011057667,0.0009864931,0.0019020222,0.0035451471,0.0016853719,0.0010160011,0.00028598378],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022820606,0.0002551645,0.013890726,0.00029526962,0.00017117111,0.000888058,0.00066737965,0.6233776,0.072757974,0.03335959,0.0010374477,0.25307137],"study_design_scores_gemma":[0.000011585986,0.00019961548,0.0012642258,0.000033313692,0.000036054167,0.00024747633,0.0000752993,0.96643084,0.021853924,0.008781507,0.0010378214,0.00002826965],"about_ca_topic_score_codex":0.0017708527,"about_ca_topic_score_gemma":0.001156603,"teacher_disagreement_score":0.0027002709,"about_ca_system_score_codex":0.0010316126,"about_ca_system_score_gemma":0.0008309124,"threshold_uncertainty_score":0.014280558},"labels":[],"label_agreement":null},{"id":"W4391974543","doi":"10.1109/tse.2024.3368208","title":"Software Testing With Large Language Models: Survey, Landscape, and Vision","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":394,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Software engineering; Software","score_opus":0.017120406674413505,"score_gpt":0.2519216021540697,"score_spread":0.23480119547965622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391974543","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035197347,0.45945078,0.44265267,0.03379083,0.00069568213,0.00033082167,0.000978464,0.00620898,0.02069431],"genre_scores_gemma":[0.42553413,0.33147672,0.22092924,0.008213248,0.0031031775,0.00058278657,0.0036683679,0.0018988195,0.004593555],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9900885,0.004911423,0.0006522069,0.0011580909,0.002961443,0.00022828947],"domain_scores_gemma":[0.9138474,0.07639063,0.0020218669,0.0029828313,0.0042288885,0.0005283364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009385707,0.0014915407,0.0018486273,0.0059925504,0.00043090034,0.0044823247,0.003635035,0.0021417132,0.0023850368],"category_scores_gemma":[0.06181108,0.0010074903,0.0015199229,0.005038078,0.003055187,0.010523414,0.0024927347,0.003093329,0.0011034748],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014588295,0.00020581722,0.010468917,0.003307177,0.00017093924,0.00019619561,0.0005721956,0.016502919,0.0010086969,0.02595612,0.013219993,0.9282451],"study_design_scores_gemma":[0.00011815417,0.00087011606,0.014190343,0.00932717,0.00046327608,0.0029726354,0.002998374,0.48164672,0.008588703,0.18447734,0.29400706,0.00034007456],"about_ca_topic_score_codex":0.0063318913,"about_ca_topic_score_gemma":0.0043580844,"teacher_disagreement_score":0.009385707,"about_ca_system_score_codex":0.002108319,"about_ca_system_score_gemma":0.0029164501,"threshold_uncertainty_score":0.04963696},"labels":[],"label_agreement":null},{"id":"W4391998081","doi":"10.1007/s10515-024-00417-0","title":"Using data mining techniques to generate test cases from graph transformation systems specifications","year":2024,"lang":"en","type":"article","venue":"Automated Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Transformation (genetics); Graph; Data mining; Test (biology); Graph rewriting; Theoretical computer science; Geology; Chemistry","score_opus":0.10704843451474054,"score_gpt":0.30585336290967075,"score_spread":0.1988049283949302,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391998081","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22921616,0.0002909192,0.7436543,0.0006383536,0.000050242343,0.0008389011,0.0049523334,0.016319687,0.004038979],"genre_scores_gemma":[0.55852854,0.0001696235,0.42978027,0.00012107794,0.000015335703,0.00044808348,0.009244152,0.0005988148,0.0010941615],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99815935,0.00045007357,0.00023194376,0.00029962757,0.00076718477,0.00009194605],"domain_scores_gemma":[0.983621,0.011780531,0.0010865519,0.0013700484,0.0019804598,0.00016145685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014804609,0.0010994959,0.00048148748,0.005504542,0.0004053357,0.0011960042,0.0013727308,0.0007787325,0.0018716841],"category_scores_gemma":[0.01624876,0.0004329741,0.0011450638,0.0024717576,0.0004086905,0.0011743216,0.0006301999,0.0008522649,0.00057158695],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006376826,0.0012258881,0.044388127,0.0011487793,0.00035683395,0.0026126816,0.0007783292,0.16864872,0.033713825,0.011187704,0.0094872415,0.7258142],"study_design_scores_gemma":[0.000120110424,0.0002959972,0.0048280736,0.00011601036,0.00014827853,0.00085422926,0.00024253715,0.936102,0.03979186,0.012980143,0.004484408,0.000036468235],"about_ca_topic_score_codex":0.0043200874,"about_ca_topic_score_gemma":0.0072358255,"teacher_disagreement_score":0.005504542,"about_ca_system_score_codex":0.0007272256,"about_ca_system_score_gemma":0.0013132792,"threshold_uncertainty_score":0.008589864},"labels":[],"label_agreement":null},{"id":"W4391998095","doi":"10.1007/s10664-023-10433-5","title":"Evaluating the impact of flaky simulators on testing autonomous driving systems","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test (biology); Replication (statistics); Key (lock); Scope (computer science); Code coverage; Simulation; Machine learning; Operating system; Statistics; Software; Mathematics; Programming language","score_opus":0.08166117695653764,"score_gpt":0.382740428437695,"score_spread":0.3010792514811574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391998095","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99214363,0.00009029847,0.006487836,0.00007151493,0.000010248334,0.000053098927,0.00006417737,0.00018936192,0.00088982336],"genre_scores_gemma":[0.9934149,0.000027934047,0.0063054026,0.00001136169,0.0000020308896,0.000014734084,0.00005856738,0.000017168848,0.00014796472],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949168,0.0033146874,0.0002769813,0.0003475549,0.0008796799,0.00026438714],"domain_scores_gemma":[0.8266977,0.15567952,0.0047927215,0.0060870154,0.0053761364,0.0013670385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049743736,0.0008984674,0.0002683462,0.0011461416,0.0003428161,0.0005359555,0.0014810615,0.0010144076,0.0013395605],"category_scores_gemma":[0.084990814,0.00043974924,0.00033722742,0.0007095672,0.00084326364,0.0016840782,0.0007400524,0.0007697117,0.00013683004],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004246696,0.006938372,0.09320131,0.0005485198,0.00034047893,0.00031174027,0.0011181594,0.6489603,0.021025112,0.0026877995,0.0009309366,0.21969053],"study_design_scores_gemma":[0.0004684874,0.010175441,0.035986368,0.000088847206,0.0002204018,0.00018555214,0.0007044156,0.9236166,0.02551823,0.0022080604,0.00076826167,0.000059320566],"about_ca_topic_score_codex":0.0073980405,"about_ca_topic_score_gemma":0.010111958,"teacher_disagreement_score":0.0073980405,"about_ca_system_score_codex":0.0012165038,"about_ca_system_score_gemma":0.0012222741,"threshold_uncertainty_score":0.026307344},"labels":[],"label_agreement":null},{"id":"W4392942125","doi":"10.1109/bcd57833.2023.10466329","title":"PyTPU: Migration of Python Code for Heterogenous Acceleration with Automated Test Generation","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Python (programming language); Computer science; Programming language; Unit testing; Software","score_opus":0.06133012185637991,"score_gpt":0.30122957876112694,"score_spread":0.239899456904747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392942125","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040422082,0.0002097659,0.47538328,0.00037912585,0.00016833999,0.00062861183,0.0015928143,0.47733766,0.0038782759],"genre_scores_gemma":[0.38927525,0.00028124778,0.51783633,0.0009085653,0.00009288849,0.0014314104,0.008796443,0.075535834,0.005842026],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9948474,0.001332136,0.00057395076,0.0009832612,0.0016682306,0.0005949533],"domain_scores_gemma":[0.9888313,0.0037325409,0.0011781792,0.004307246,0.001539936,0.00041076573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005196527,0.0021491426,0.00077747344,0.0020916455,0.0008210388,0.0018569732,0.004293623,0.0014130701,0.008005891],"category_scores_gemma":[0.022115815,0.0016808537,0.0021371362,0.000864766,0.0026563292,0.003652183,0.004211045,0.0027858247,0.0041603274],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002717599,0.001215156,0.048570946,0.002156348,0.0006023823,0.0038028876,0.0036803514,0.072332025,0.08021352,0.029059635,0.12621628,0.62943286],"study_design_scores_gemma":[0.0005486791,0.0009378042,0.013720659,0.00056818756,0.00020013511,0.001985944,0.0005041358,0.64858353,0.19777791,0.031488825,0.10323779,0.00044633338],"about_ca_topic_score_codex":0.0039143325,"about_ca_topic_score_gemma":0.0030341763,"teacher_disagreement_score":0.008005891,"about_ca_system_score_codex":0.0012566934,"about_ca_system_score_gemma":0.0040203067,"threshold_uncertainty_score":0.027482212},"labels":[],"label_agreement":null},{"id":"W4393108609","doi":"10.1145/3589335.3651463","title":"A Study of Vulnerability Repair in JavaScript Programs with Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"JavaScript; Computer science; Unobtrusive JavaScript; Secure coding; Web application; Context (archaeology); Programming language; World Wide Web; Computer security; Vulnerability (computing); Code (set theory); Software engineering; Rich Internet application; Software security assurance; Information security","score_opus":0.04482670237209341,"score_gpt":0.312129429966993,"score_spread":0.2673027275948996,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393108609","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92389816,0.00042827646,0.07293482,0.0003043679,0.000012425241,0.00005684365,0.00008282226,0.0012323887,0.0010498805],"genre_scores_gemma":[0.9607445,0.00009706362,0.038338836,0.000041921245,0.000008532174,0.000045340857,0.000095244446,0.00022778682,0.00040087098],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99766445,0.0012139165,0.00009177215,0.0003734379,0.0005056579,0.00015069814],"domain_scores_gemma":[0.928952,0.062587656,0.0035748596,0.0028110603,0.0015004723,0.00057386636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029413567,0.0005832693,0.0005163067,0.0008651513,0.000519648,0.0006786612,0.00094258547,0.00081704906,0.00064539706],"category_scores_gemma":[0.04769882,0.0004490089,0.0006452061,0.00066457235,0.00124503,0.0019221735,0.0009224091,0.0012946208,0.00010938147],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012007217,0.0012938005,0.09974703,0.0010172352,0.00027150498,0.0021785414,0.0056733256,0.652375,0.06451317,0.025333464,0.0027272925,0.14366901],"study_design_scores_gemma":[0.00003843099,0.0003174118,0.007532209,0.000029883962,0.00004839465,0.00031769494,0.00031006368,0.9729782,0.011843937,0.0057922495,0.0007668626,0.000024766836],"about_ca_topic_score_codex":0.004781592,"about_ca_topic_score_gemma":0.004156299,"teacher_disagreement_score":0.004781592,"about_ca_system_score_codex":0.0010701413,"about_ca_system_score_gemma":0.00089368527,"threshold_uncertainty_score":0.015555561},"labels":[],"label_agreement":null},{"id":"W4393547037","doi":"10.5281/zenodo.2677940","title":"Testing Selectivity of USP5 Zf-UBD Analogues with SPR Assay","year":2019,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Selectivity; Chemistry; Mathematics; Chromatography; Computational biology; Biological system; Computer science; Biology; Biochemistry","score_opus":0.05905873052827128,"score_gpt":0.26357634429834026,"score_spread":0.204517613770069,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393547037","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045273425,0.00055678794,0.0003743954,0.0001596987,0.000060606522,0.00006918116,0.9920204,0.00080748263,0.0014240028],"genre_scores_gemma":[0.0039497656,0.00016245533,0.0010623729,0.00010606614,0.0000062224844,0.0001788468,0.9935395,0.00008551083,0.00090923003],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99812895,0.00034913202,0.00021613063,0.0006187527,0.00047832844,0.00020874133],"domain_scores_gemma":[0.9980478,0.0007139275,0.000241771,0.00048184546,0.0003858471,0.00012883956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016775036,0.0032202748,0.0017207494,0.002870267,0.00071540836,0.0015989039,0.002543947,0.0024918576,0.014402513],"category_scores_gemma":[0.0047109094,0.00048953015,0.0017202097,0.0033965784,0.0005529446,0.00067492883,0.0012771143,0.0018007958,0.018480824],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015453148,0.00064741017,0.008575631,0.005502459,0.00053770584,0.00024997094,0.00007056674,0.0095419735,0.0040809535,0.0014943291,0.9473562,0.0203975],"study_design_scores_gemma":[0.0026814174,0.00055818894,0.016531844,0.0005762586,0.0005335928,0.0005123234,0.00017135091,0.008985315,0.010574327,0.0029015646,0.9558678,0.00010597132],"about_ca_topic_score_codex":0.014094982,"about_ca_topic_score_gemma":0.021433016,"teacher_disagreement_score":0.014402513,"about_ca_system_score_codex":0.0013243518,"about_ca_system_score_gemma":0.0016989933,"threshold_uncertainty_score":0.048181176},"labels":[],"label_agreement":null},{"id":"W4393610751","doi":"10.5281/zenodo.6994691","title":"Flakify: A Black-Box, Language Model-based Predictor for Flaky Tests – Replication Package","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Replication (statistics); Black box; Computer science; Programming language; Mathematics; Artificial intelligence; Statistics","score_opus":0.03671388466794158,"score_gpt":0.283911857080683,"score_spread":0.24719797241274144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393610751","genre_codex":"software","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037711358,0.00041548334,0.15008433,0.0010610655,0.0006406014,0.0005937295,0.19576871,0.64101183,0.006653112],"genre_scores_gemma":[0.039226536,0.00038409326,0.23202518,0.0018422022,0.00035390133,0.003493964,0.60408014,0.10131465,0.017279403],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9967158,0.0007372918,0.00032866639,0.0010041096,0.00091120345,0.00030300237],"domain_scores_gemma":[0.9890416,0.005856366,0.0005757049,0.0021498748,0.0019718148,0.00040463352],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005490847,0.0033133235,0.001557373,0.0028125271,0.0007954733,0.00351736,0.0035801257,0.001534546,0.09419644],"category_scores_gemma":[0.041406516,0.0014179831,0.0027119417,0.001534007,0.00070601655,0.0040886113,0.0035074693,0.003668678,0.109958924],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006203468,0.000105562685,0.0062342933,0.00065633655,0.00016139766,0.00015359398,0.00009991444,0.007923801,0.0014987673,0.0024581316,0.8948949,0.085192904],"study_design_scores_gemma":[0.0010571613,0.0004010634,0.009106966,0.00077317975,0.00023749963,0.00046957907,0.00027344617,0.26701885,0.02112242,0.048698824,0.6505524,0.00028860607],"about_ca_topic_score_codex":0.0074490705,"about_ca_topic_score_gemma":0.009320468,"teacher_disagreement_score":0.09419644,"about_ca_system_score_codex":0.0010613378,"about_ca_system_score_gemma":0.004369263,"threshold_uncertainty_score":0.3151185},"labels":[],"label_agreement":null},{"id":"W4393662211","doi":"10.5281/zenodo.2677941","title":"Testing Selectivity of USP5 Zf-UBD Analogues with SPR Assay","year":2019,"lang":"en","type":"dataset","venue":"Figshare","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Selectivity; Chemistry; Computer science; Biochemistry","score_opus":0.07578319720065074,"score_gpt":0.28229883875497114,"score_spread":0.2065156415543204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393662211","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029344151,0.0005108451,0.0002531884,0.00014594092,0.000054253193,0.00008482238,0.99403197,0.0007694725,0.001215014],"genre_scores_gemma":[0.004060982,0.00020581712,0.0012379816,0.000162489,0.000006600314,0.00030552584,0.99311453,0.000110612644,0.0007954073],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99814,0.00034258695,0.00022698747,0.0006360526,0.00045421853,0.00020018645],"domain_scores_gemma":[0.9976921,0.0009903008,0.00026035943,0.00049342465,0.0004330568,0.00013075936],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020142414,0.0030985388,0.0020562676,0.0028550324,0.000872582,0.001812255,0.002957779,0.0025363981,0.022371216],"category_scores_gemma":[0.0065840087,0.0006319977,0.0020263572,0.003769915,0.0005801956,0.00079753855,0.0014215993,0.0018818284,0.018648231],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017478142,0.0006610162,0.009582696,0.008683025,0.0007662176,0.00019999365,0.00007314957,0.008748878,0.0036054363,0.0014546178,0.94444025,0.020036917],"study_design_scores_gemma":[0.0044218493,0.0006399387,0.013822089,0.0007776516,0.0008554327,0.00038894147,0.00017703245,0.0076860855,0.009845416,0.0029947823,0.9582591,0.00013167958],"about_ca_topic_score_codex":0.014829816,"about_ca_topic_score_gemma":0.024559205,"teacher_disagreement_score":0.022371216,"about_ca_system_score_codex":0.001374669,"about_ca_system_score_gemma":0.0021188154,"threshold_uncertainty_score":0.074839115},"labels":[],"label_agreement":null},{"id":"W4393677065","doi":"10.5281/zenodo.7455766","title":"ATM: Black-box Test Case Minimization based on Test Code Similarity and Evolutionary Search – Replication Package","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Replication (statistics); Computer science; Code (set theory); Test (biology); Similarity (geometry); Programming language; Artificial intelligence; Biology; Mathematics; Statistics","score_opus":0.04402094876027669,"score_gpt":0.2760535441570954,"score_spread":0.2320325953968187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393677065","genre_codex":"methods","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029507808,0.00013047707,0.74234504,0.00043711744,0.00024432276,0.0007443731,0.008870228,0.2331686,0.011109049],"genre_scores_gemma":[0.07439568,0.00020735382,0.7946454,0.0006376796,0.0002801591,0.002860753,0.04116856,0.061451536,0.024352813],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99515826,0.0011276914,0.0003959449,0.000689825,0.0023865874,0.00024163286],"domain_scores_gemma":[0.9857567,0.00561246,0.00077796844,0.0041686655,0.0033887164,0.00029554215],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054724733,0.0021602435,0.0011092959,0.0025182157,0.0005542057,0.0019819443,0.004015624,0.0013659403,0.105068274],"category_scores_gemma":[0.034656398,0.0014368648,0.0019597309,0.001514829,0.00060715456,0.002493591,0.002391607,0.0019602233,0.049022984],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011908342,0.00025146967,0.0029293722,0.0006865602,0.00021521485,0.00031258917,0.00014044758,0.06876968,0.011164059,0.022445465,0.4984357,0.39345863],"study_design_scores_gemma":[0.00074634637,0.00051618874,0.0025171756,0.00019545494,0.00011321647,0.00060008484,0.000056242716,0.7520556,0.027840098,0.041796263,0.17339131,0.0001719208],"about_ca_topic_score_codex":0.0030593218,"about_ca_topic_score_gemma":0.0020214713,"teacher_disagreement_score":0.105068274,"about_ca_system_score_codex":0.0008705379,"about_ca_system_score_gemma":0.0022891217,"threshold_uncertainty_score":0.35148835},"labels":[],"label_agreement":null},{"id":"W4393740131","doi":"10.5281/zenodo.6994692","title":"Flakify: A Black-Box, Language Model-based Predictor for Flaky Tests – Replication Package","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Replication (statistics); Black box; Computer science; R package; Programming language; Artificial intelligence; Mathematics; Statistics","score_opus":0.03671388466794158,"score_gpt":0.283911857080683,"score_spread":0.24719797241274144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393740131","genre_codex":"software","genre_gemma":"dataset","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037711358,0.00041548334,0.15008433,0.0010610655,0.0006406014,0.0005937295,0.19576871,0.64101183,0.006653112],"genre_scores_gemma":[0.039226536,0.00038409326,0.23202518,0.0018422022,0.00035390133,0.003493964,0.60408014,0.10131465,0.017279403],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9967158,0.0007372918,0.00032866639,0.0010041096,0.00091120345,0.00030300237],"domain_scores_gemma":[0.9890416,0.005856366,0.0005757049,0.0021498748,0.0019718148,0.00040463352],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.005490847,0.0033133235,0.001557373,0.0028125271,0.0007954733,0.00351736,0.0035801257,0.001534546,0.09419644],"category_scores_gemma":[0.041406516,0.0014179831,0.0027119417,0.001534007,0.00070601655,0.0040886113,0.0035074693,0.003668678,0.109958924],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006203468,0.000105562685,0.0062342933,0.00065633655,0.00016139766,0.00015359398,0.00009991444,0.007923801,0.0014987673,0.0024581316,0.8948949,0.085192904],"study_design_scores_gemma":[0.0010571613,0.0004010634,0.009106966,0.00077317975,0.00023749963,0.00046957907,0.00027344617,0.26701885,0.02112242,0.048698824,0.6505524,0.00028860607],"about_ca_topic_score_codex":0.0074490705,"about_ca_topic_score_gemma":0.009320468,"teacher_disagreement_score":0.99450916,"about_ca_system_score_codex":0.0010613378,"about_ca_system_score_gemma":0.004369263,"threshold_uncertainty_score":0.3151185},"labels":[],"label_agreement":null},{"id":"W4393757241","doi":"10.5281/zenodo.7455765","title":"ATM: Black-box Test Case Minimization based on Test Code Similarity and Evolutionary Search – Replication Package","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Replication (statistics); Computer science; Code (set theory); Test (biology); Similarity (geometry); Minification; Programming language; Artificial intelligence; Biology; Mathematics; Statistics; Paleontology","score_opus":0.04402094876027669,"score_gpt":0.2760535441570954,"score_spread":0.2320325953968187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393757241","genre_codex":"methods","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029507808,0.00013047707,0.74234504,0.00043711744,0.00024432276,0.0007443731,0.008870228,0.2331686,0.011109049],"genre_scores_gemma":[0.07439568,0.00020735382,0.7946454,0.0006376796,0.0002801591,0.002860753,0.04116856,0.061451536,0.024352813],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99515826,0.0011276914,0.0003959449,0.000689825,0.0023865874,0.00024163286],"domain_scores_gemma":[0.9857567,0.00561246,0.00077796844,0.0041686655,0.0033887164,0.00029554215],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054724733,0.0021602435,0.0011092959,0.0025182157,0.0005542057,0.0019819443,0.004015624,0.0013659403,0.105068274],"category_scores_gemma":[0.034656398,0.0014368648,0.0019597309,0.001514829,0.00060715456,0.002493591,0.002391607,0.0019602233,0.049022984],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011908342,0.00025146967,0.0029293722,0.0006865602,0.00021521485,0.00031258917,0.00014044758,0.06876968,0.011164059,0.022445465,0.4984357,0.39345863],"study_design_scores_gemma":[0.00074634637,0.00051618874,0.0025171756,0.00019545494,0.00011321647,0.00060008484,0.000056242716,0.7520556,0.027840098,0.041796263,0.17339131,0.0001719208],"about_ca_topic_score_codex":0.0030593218,"about_ca_topic_score_gemma":0.0020214713,"teacher_disagreement_score":0.105068274,"about_ca_system_score_codex":0.0008705379,"about_ca_system_score_gemma":0.0022891217,"threshold_uncertainty_score":0.35148835},"labels":[],"label_agreement":null},{"id":"W4393772048","doi":"10.5281/zenodo.8309515","title":"Search-based Software Testing Driven by Automatically Generated and Manually Defined Fitness Functions - Dataset and Results","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Software; Software testing; Data mining; Machine learning; Artificial intelligence; Programming language","score_opus":0.053576618488732214,"score_gpt":0.26812928173125283,"score_spread":0.21455266324252062,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393772048","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019979756,0.00023393155,0.0005292339,0.00008752879,0.000038223327,0.000049734186,0.9935183,0.002508364,0.0010367763],"genre_scores_gemma":[0.0017503656,0.000054926524,0.0010001246,0.000047497873,0.000004777766,0.00013817273,0.99646616,0.00015522339,0.00038277797],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9969805,0.00065351615,0.00039644033,0.00079099333,0.00090887013,0.0002697094],"domain_scores_gemma":[0.9949976,0.0018775119,0.0003625284,0.0015272173,0.0010057315,0.00022950761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020660795,0.0036519524,0.0016861872,0.0034793615,0.00066033384,0.0017245165,0.0030462602,0.0029553797,0.015784057],"category_scores_gemma":[0.008993356,0.0006497366,0.002208036,0.0037523946,0.0004931373,0.0008970714,0.0016827936,0.0017576923,0.029670741],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005531675,0.00025262046,0.0044623283,0.002860573,0.00022767967,0.00011547843,0.000046584944,0.007529562,0.0008108404,0.000799961,0.96668595,0.015655218],"study_design_scores_gemma":[0.0020786335,0.0003771276,0.031249044,0.0012856788,0.00038297696,0.0005275147,0.00021373975,0.025386008,0.007133895,0.0057054865,0.92544967,0.00021027596],"about_ca_topic_score_codex":0.014230123,"about_ca_topic_score_gemma":0.02630334,"teacher_disagreement_score":0.015784057,"about_ca_system_score_codex":0.0017494357,"about_ca_system_score_gemma":0.0017467238,"threshold_uncertainty_score":0.05280292},"labels":[],"label_agreement":null},{"id":"W4393978605","doi":"10.1007/978-3-031-57259-3_19","title":"TracerX: Pruning Dynamic Symbolic Execution with Deletion and Weakest Precondition Interpolation (Competition Contribution)","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Precondition; Computer science; Pruning; Interpolation (computer graphics); Competition (biology); Algorithm; Theoretical computer science; Artificial intelligence; Programming language","score_opus":0.007317886509596298,"score_gpt":0.2327990402987306,"score_spread":0.2254811537891343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393978605","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038584463,0.00036339133,0.9476842,0.00030474178,0.000109638226,0.00011951047,0.0002633151,0.0089678,0.0036029255],"genre_scores_gemma":[0.43067926,0.0001807775,0.5600697,0.00020561076,0.00005202411,0.00021434443,0.0009886113,0.0012735659,0.0063360664],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987698,0.0003267873,0.00005676638,0.00017559923,0.00053596817,0.0001349033],"domain_scores_gemma":[0.9973061,0.0016216667,0.00013835503,0.00046116454,0.00034198991,0.00013057159],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015087568,0.0008431588,0.00088900997,0.0008760214,0.00048156202,0.0008509569,0.0021664784,0.0009309239,0.008253421],"category_scores_gemma":[0.0054991255,0.0004509277,0.00080984033,0.0007221196,0.0015259824,0.0012998409,0.0022397235,0.0016769327,0.0009227735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011067706,0.0002623301,0.0036832434,0.00063140364,0.00009098982,0.00055062183,0.0003017967,0.48723885,0.020233769,0.0812505,0.016589763,0.3880599],"study_design_scores_gemma":[0.000056024688,0.0001250834,0.00016760819,0.000028934906,0.000016033044,0.00006922349,0.000020864978,0.96602803,0.009451067,0.020121811,0.0039042912,0.000011018541],"about_ca_topic_score_codex":0.004379687,"about_ca_topic_score_gemma":0.0049258354,"teacher_disagreement_score":0.008253421,"about_ca_system_score_codex":0.0009014266,"about_ca_system_score_gemma":0.0023280568,"threshold_uncertainty_score":0.02761048},"labels":[],"label_agreement":null},{"id":"W4394015350","doi":"10.1016/j.infsof.2024.107468","title":"Effective test generation using pre-trained Large Language Models and mutation testing","year":2024,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":97,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Test (biology); Mutation; Computer science; Natural language processing; Genetics; Biology; Gene","score_opus":0.01684748034253352,"score_gpt":0.2775876674362418,"score_spread":0.2607401870937083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394015350","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18236811,0.0009965552,0.78578335,0.000658948,0.00012644527,0.00031312072,0.00063553697,0.027165594,0.0019522457],"genre_scores_gemma":[0.7050749,0.00019412453,0.2882021,0.0007379483,0.00006047526,0.00034418554,0.0029245852,0.0006687777,0.001792911],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99867886,0.00041450214,0.0000696754,0.00042465443,0.00029711775,0.00011515063],"domain_scores_gemma":[0.9941551,0.004301644,0.00034450812,0.00044816613,0.0005950739,0.00015548492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012374524,0.0015793105,0.0010163524,0.0013362799,0.00035954942,0.0007329915,0.0018844957,0.0015193701,0.0017270212],"category_scores_gemma":[0.008385114,0.00048273863,0.0011783094,0.00060415606,0.0006916634,0.0012813867,0.0010020516,0.001509285,0.0008576497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046000004,0.0006600637,0.008439554,0.0004469495,0.00017370065,0.0012824659,0.000273793,0.39253452,0.03624671,0.0020628003,0.008417867,0.54900163],"study_design_scores_gemma":[0.000033072458,0.00009789668,0.0005333508,0.000013551543,0.000023673403,0.0001596766,0.000028716564,0.9910425,0.0058751707,0.0015927135,0.0005876768,0.000012000323],"about_ca_topic_score_codex":0.0049643717,"about_ca_topic_score_gemma":0.006390506,"teacher_disagreement_score":0.0049643717,"about_ca_system_score_codex":0.0008024166,"about_ca_system_score_gemma":0.001382725,"threshold_uncertainty_score":0.009870946},"labels":[],"label_agreement":null},{"id":"W4394746056","doi":"10.1145/3597503.3639112","title":"Mozi: Discovering DBMS Bugs via Configuration-Based Equivalent Transformation","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"China Postdoctoral Science Foundation; National Natural Science Foundation of China","keywords":"Computer science; Correctness; Programming language; SQL; Semantics (computer science); Database; Query language; Transformation (genetics); Task (project management)","score_opus":0.01918079054616061,"score_gpt":0.2707753826562442,"score_spread":0.25159459211008356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394746056","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0646181,0.0004504231,0.88932776,0.00059572206,0.0000748407,0.00040757333,0.0006860784,0.041399725,0.0024398793],"genre_scores_gemma":[0.36944732,0.00017985197,0.6244805,0.00040427502,0.00003877661,0.00035109185,0.0019960287,0.0019908445,0.0011112863],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9865384,0.0027983976,0.00095737685,0.0031020846,0.0053877393,0.0012160119],"domain_scores_gemma":[0.9787416,0.008456806,0.0034285013,0.0063036946,0.002564009,0.00050533755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069161383,0.0024635165,0.001856618,0.005259435,0.0012639186,0.003567623,0.005723133,0.0026842095,0.0024340684],"category_scores_gemma":[0.044931695,0.0014288084,0.0027385342,0.0023111545,0.0042120344,0.006795346,0.005214926,0.0031456712,0.00077187334],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018736323,0.0014802497,0.1233657,0.0014972029,0.0007910038,0.0034824242,0.0032005834,0.20435908,0.07110326,0.12714238,0.019059211,0.4426452],"study_design_scores_gemma":[0.000141854,0.00050704734,0.005878457,0.00015442856,0.00019066127,0.0013023395,0.00051944086,0.8583049,0.03720901,0.08875596,0.006850495,0.00018538516],"about_ca_topic_score_codex":0.006683354,"about_ca_topic_score_gemma":0.006636204,"teacher_disagreement_score":0.0069161383,"about_ca_system_score_codex":0.0018682283,"about_ca_system_score_gemma":0.0029919385,"threshold_uncertainty_score":0.03657645},"labels":[],"label_agreement":null},{"id":"W4394927149","doi":"10.1007/s10270-024-01170-4","title":"Improving repair of semantic ATL errors using a social diversity metric","year":2024,"lang":"en","type":"article","venue":"Software & Systems Modeling","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; McGill University; Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Metric (unit); Diversity (politics); Natural language processing; Information retrieval; Data science; Artificial intelligence; Sociology","score_opus":0.055793967027818005,"score_gpt":0.2822911530338823,"score_spread":0.2264971860060643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394927149","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55373925,0.0006902659,0.43542936,0.00060129055,0.00012800336,0.0001604591,0.0003078475,0.0034583376,0.0054851635],"genre_scores_gemma":[0.9231866,0.00007396557,0.07523319,0.00006240708,0.000051364637,0.000036011654,0.0003309668,0.00016665937,0.000858782],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99030775,0.0030697729,0.0006532391,0.0011024935,0.0042659556,0.0006007965],"domain_scores_gemma":[0.96269816,0.015681164,0.0051284875,0.0066512213,0.008387726,0.0014531738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005366365,0.0010146942,0.0012840654,0.0046673617,0.0015634014,0.001588762,0.0018758435,0.0014959829,0.0017311588],"category_scores_gemma":[0.03352158,0.0003496292,0.000795103,0.0022569795,0.0011270248,0.004970568,0.0040103504,0.0011867626,0.00039868543],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013311274,0.0013300456,0.06775519,0.00042415806,0.00043377982,0.0005345328,0.0014393262,0.22217192,0.046856202,0.0132934125,0.0045584855,0.6398718],"study_design_scores_gemma":[0.00009466495,0.0012404817,0.014324967,0.000058231344,0.00021992823,0.00040393264,0.0009154584,0.93006253,0.026277564,0.02324528,0.0030827527,0.00007423014],"about_ca_topic_score_codex":0.0024147124,"about_ca_topic_score_gemma":0.0054845023,"teacher_disagreement_score":0.005366365,"about_ca_system_score_codex":0.0011183514,"about_ca_system_score_gemma":0.0017645822,"threshold_uncertainty_score":0.028380394},"labels":[],"label_agreement":null},{"id":"W4395113782","doi":"10.1007/978-3-031-57540-2_5","title":"Accurify: Automated New Testflows Generation for Attack Variants in Threat Hunting","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Ericsson (Canada)","funders":"","keywords":"Computer science; Computer security; Artificial intelligence","score_opus":0.056137756697451434,"score_gpt":0.3139452280827616,"score_spread":0.2578074713853102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4395113782","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0303039,0.00040074086,0.8404602,0.00021324855,0.00019295613,0.00019963789,0.0013081321,0.1204309,0.006490352],"genre_scores_gemma":[0.33824602,0.00021394451,0.6375742,0.00025989872,0.00009660725,0.00026385736,0.0039038775,0.010856698,0.008584989],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99881434,0.00024821097,0.00008175487,0.00030664058,0.00042720343,0.00012181649],"domain_scores_gemma":[0.9971615,0.001532458,0.00018335217,0.00068180135,0.00034005175,0.00010092995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011597117,0.0017958867,0.00075215276,0.0015878282,0.0004768326,0.0013551151,0.0025343783,0.0013723968,0.014965926],"category_scores_gemma":[0.005068708,0.0007924972,0.0012338338,0.0007165133,0.00082154834,0.0025094373,0.0015727549,0.0013567661,0.0037078585],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010387308,0.00034359712,0.006393948,0.0005638497,0.0002250852,0.0007981524,0.00050407194,0.06623258,0.050415833,0.016640192,0.06214319,0.7947008],"study_design_scores_gemma":[0.00025278484,0.00034258718,0.0021076477,0.00013196682,0.00015422453,0.0010014066,0.000103641025,0.8319029,0.09336865,0.032805257,0.037720867,0.000108036555],"about_ca_topic_score_codex":0.0018342129,"about_ca_topic_score_gemma":0.0024245866,"teacher_disagreement_score":0.014965926,"about_ca_system_score_codex":0.00068171404,"about_ca_system_score_gemma":0.00071341655,"threshold_uncertainty_score":0.050065994},"labels":[],"label_agreement":null},{"id":"W4396214523","doi":"10.1145/3649850","title":"PyDex: Repairing Bugs in Introductory Python Assignments using LLMs","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Python (programming language); Programming language; Computer science; Software engineering","score_opus":0.022964714500130406,"score_gpt":0.3016022634851895,"score_spread":0.2786375489850591,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396214523","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16909966,0.00045515536,0.14802052,0.0005756186,0.00017234098,0.00034954865,0.0040716897,0.67272264,0.0045327903],"genre_scores_gemma":[0.6615663,0.00031396621,0.29406163,0.00075579755,0.000043662472,0.00035699148,0.011932253,0.01808441,0.012884946],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983706,0.00036280704,0.00014369898,0.00052452536,0.00047052198,0.00012794072],"domain_scores_gemma":[0.99505,0.0025090622,0.0005688918,0.001083319,0.00051137235,0.00027738692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018527127,0.0014468672,0.0006019085,0.001121522,0.0005513198,0.0009203794,0.0027502105,0.0010635372,0.008458891],"category_scores_gemma":[0.011796616,0.0007182306,0.0005741901,0.00061711797,0.000977674,0.0034487182,0.0028852993,0.001554242,0.0045894175],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019990627,0.0009261215,0.02871657,0.0019327395,0.00013677825,0.0015611254,0.0042902133,0.018215349,0.042736493,0.003203007,0.124180555,0.77210194],"study_design_scores_gemma":[0.00061972346,0.0022084387,0.030912723,0.00043848436,0.00021590068,0.0025769866,0.0026584913,0.5059935,0.2667195,0.013860048,0.17342722,0.0003690701],"about_ca_topic_score_codex":0.002523351,"about_ca_topic_score_gemma":0.0039135837,"teacher_disagreement_score":0.008458891,"about_ca_system_score_codex":0.0007757277,"about_ca_system_score_gemma":0.0012825619,"threshold_uncertainty_score":0.028297782},"labels":[],"label_agreement":null},{"id":"W4396796398","doi":"10.3390/biomedinformatics4020069","title":"A Smartphone-Based Algorithm for L Test Subtask Segmentation","year":2024,"lang":"en","type":"article","venue":"BioMedInformatics","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test (biology); Artificial intelligence; Segmentation; Algorithm; Computer vision; Geology","score_opus":0.019163561674323558,"score_gpt":0.28097800109259485,"score_spread":0.2618144394182713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396796398","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04258592,0.00040302466,0.93728447,0.0002610935,0.000104210405,0.00052335433,0.0009332585,0.015660673,0.002243997],"genre_scores_gemma":[0.23519625,0.00017148274,0.75841737,0.0002299297,0.000049772236,0.00074424123,0.001701034,0.000462137,0.003027741],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992805,0.000107530694,0.000090073896,0.000253786,0.00021269078,0.00005551705],"domain_scores_gemma":[0.99804544,0.00070247403,0.0002063205,0.00014958734,0.0007970512,0.000099146906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006893254,0.0012165577,0.000740777,0.0019717913,0.00043414478,0.0009857927,0.0009965668,0.00092704437,0.00528592],"category_scores_gemma":[0.004621288,0.00030357135,0.00050744327,0.00080849294,0.00025067767,0.00049868843,0.0008204831,0.0006041028,0.0032925976],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006935214,0.00017709183,0.0095185265,0.00017438369,0.000082711354,0.00028130107,0.00019788134,0.01205036,0.03346172,0.00087481947,0.0134766195,0.9290111],"study_design_scores_gemma":[0.00024458204,0.00046976365,0.017201358,0.0000799553,0.000091856484,0.0012119546,0.00019054099,0.91237676,0.048601884,0.0037382492,0.015699565,0.00009358282],"about_ca_topic_score_codex":0.0056489534,"about_ca_topic_score_gemma":0.007509106,"teacher_disagreement_score":0.0056489534,"about_ca_system_score_codex":0.0005940874,"about_ca_system_score_gemma":0.0011902461,"threshold_uncertainty_score":0.017683148},"labels":[],"label_agreement":null},{"id":"W4396871671","doi":"10.1145/3664601","title":"Testing Updated Apps by Adapting Learned Models","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Centre National de la Recherche Scientifique","keywords":"Computer science; Correctness; Adaptation (eye); Software; Software inspection; Software performance testing; Machine learning; Software engineering; Human–computer interaction; Software development; Software quality; Operating system; Software construction; Programming language","score_opus":0.1504107154579445,"score_gpt":0.32533558583462185,"score_spread":0.17492487037667737,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396871671","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3614489,0.0016958071,0.5704099,0.0007734447,0.0002379842,0.00066036556,0.0015069564,0.057163887,0.006102831],"genre_scores_gemma":[0.772447,0.0004720799,0.21854551,0.0004923003,0.000060064605,0.00042518158,0.0025553375,0.0019062405,0.0030963183],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9936081,0.0016730445,0.00036164344,0.0016257791,0.0023644639,0.00036702715],"domain_scores_gemma":[0.9754047,0.011732799,0.0015686437,0.007786157,0.003073535,0.00043409705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028974065,0.0025087127,0.0009318402,0.0014849557,0.00034999187,0.0017243244,0.0043767644,0.0014997227,0.002371946],"category_scores_gemma":[0.038369454,0.0012579897,0.0012146246,0.0007159737,0.0008685481,0.004483138,0.0030872915,0.0026229313,0.0014015343],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010268135,0.0011238393,0.0422955,0.00076854625,0.00033207992,0.00063784997,0.0008145503,0.25262776,0.03093397,0.0025280418,0.008327145,0.65858394],"study_design_scores_gemma":[0.00006108849,0.00031231984,0.0030061281,0.00007535555,0.000106833744,0.00026880548,0.000092469774,0.9750768,0.012505013,0.003851343,0.0045943805,0.00004948427],"about_ca_topic_score_codex":0.007080216,"about_ca_topic_score_gemma":0.011327507,"teacher_disagreement_score":0.007080216,"about_ca_system_score_codex":0.0011082771,"about_ca_system_score_gemma":0.0023426474,"threshold_uncertainty_score":0.015323162},"labels":[],"label_agreement":null},{"id":"W4398239453","doi":"10.1145/3639478.3643062","title":"Technical Brief on Software Engineering for FMware","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Software; Engineering management; Trustworthiness; Foundation (evidence); Quality (philosophy); Software engineering; Software quality; Software testing; Software development; Systems engineering; Engineering; Computer security","score_opus":0.019046383439439946,"score_gpt":0.26932644212361306,"score_spread":0.25028005868417313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398239453","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0024250492,0.16032372,0.51410246,0.06222283,0.023490287,0.00044122693,0.00086091843,0.0037193447,0.23241423],"genre_scores_gemma":[0.029155323,0.19435851,0.51513034,0.023108894,0.017886272,0.0007123154,0.0027091186,0.0021898455,0.2147494],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99781454,0.00052627275,0.0003062308,0.00025708863,0.00097083475,0.00012506878],"domain_scores_gemma":[0.99656194,0.0015015352,0.0002723208,0.00040398916,0.001014947,0.00024538377],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002415729,0.0011915019,0.0003886705,0.0024678058,0.0010649606,0.002358166,0.00089592376,0.0025878353,0.026475681],"category_scores_gemma":[0.006356245,0.00057844556,0.0007059277,0.0029415747,0.0014387224,0.005222399,0.0016892023,0.004755245,0.026151175],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017810893,0.00005231145,0.00033586248,0.001267267,0.00001438654,0.0004387647,0.00039338227,0.0010025003,0.0034494456,0.25938368,0.30841705,0.4252275],"study_design_scores_gemma":[0.0000019514628,0.0000259777,0.00015443748,0.00038550704,0.0000033818183,0.00059552456,0.000046739,0.0003894215,0.00049793586,0.034850072,0.96303743,0.000011576251],"about_ca_topic_score_codex":0.0010589452,"about_ca_topic_score_gemma":0.0018481067,"teacher_disagreement_score":0.026475681,"about_ca_system_score_codex":0.0014096837,"about_ca_system_score_gemma":0.0017159431,"threshold_uncertainty_score":0.08856994},"labels":[],"label_agreement":null},{"id":"W4399036954","doi":"10.29007/rdbb","title":"Efficient Simulation for Hardware Model Checking","year":2024,"lang":"en","type":"article","venue":"EPiC series in computing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Correctness; Executable; Model checking; Semantics (computer science); Programming language; Formal verification; Set (abstract data type); Speedup; Symbolic trajectory evaluation; Computer engineering; Parallel computing; Theoretical computer science","score_opus":0.043641389716917105,"score_gpt":0.33103879876707565,"score_spread":0.28739740905015854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399036954","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0064917053,0.00017109563,0.979795,0.00019062302,0.000057159752,0.00008706434,0.0002701724,0.008296305,0.0046408726],"genre_scores_gemma":[0.24241823,0.00045783902,0.7481264,0.00020620016,0.00005477037,0.0005494094,0.0014990348,0.002451925,0.004236292],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975979,0.00094364263,0.00015801404,0.0002713821,0.00079199544,0.00023705249],"domain_scores_gemma":[0.9962065,0.0023213718,0.00017899097,0.0008514181,0.00037769866,0.0000640187],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015741347,0.0014660927,0.0009698684,0.0008126705,0.00054163823,0.0015919995,0.0018109953,0.0010350054,0.014760561],"category_scores_gemma":[0.0076785157,0.0007782766,0.0020993263,0.0007584008,0.0012463619,0.0027613516,0.0018178758,0.0023507422,0.0026700648],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034526133,0.00014432117,0.0015766132,0.0006541754,0.00010931732,0.00025014204,0.00018028553,0.67333627,0.015997449,0.20679152,0.008060624,0.09255403],"study_design_scores_gemma":[0.000045287812,0.000038821894,0.00008053002,0.000048512604,0.000018790988,0.000049671195,0.000019068328,0.92138237,0.0092173815,0.059455097,0.009628903,0.000015566766],"about_ca_topic_score_codex":0.0032026584,"about_ca_topic_score_gemma":0.0046680504,"teacher_disagreement_score":0.014760561,"about_ca_system_score_codex":0.0014997906,"about_ca_system_score_gemma":0.0021305142,"threshold_uncertainty_score":0.04937899},"labels":[],"label_agreement":null},{"id":"W4399061142","doi":"10.1145/3663529.3663787","title":"On Polyglot Program Testing","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Agence Nationale de la Recherche","keywords":"Polyglot; Computer science; Programming language","score_opus":0.04053516960178625,"score_gpt":0.3158553354438702,"score_spread":0.2753201658420839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399061142","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022293717,0.0051258653,0.92977977,0.0051426208,0.00022909328,0.00018264142,0.00017635184,0.002990022,0.034079943],"genre_scores_gemma":[0.40457872,0.014572721,0.5459929,0.0070302,0.00070125423,0.00044445007,0.0010813194,0.0029707432,0.022627741],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98961323,0.004597544,0.00043049004,0.0012639058,0.0035196343,0.00057511654],"domain_scores_gemma":[0.9739705,0.01731472,0.0013414852,0.004410599,0.0025327336,0.00042992664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006153913,0.001231461,0.000698995,0.0028740894,0.0010335506,0.003113962,0.0019618203,0.0016509419,0.0055923695],"category_scores_gemma":[0.023812396,0.0006525273,0.0006573886,0.0035769579,0.006278881,0.008154274,0.004111872,0.0040513007,0.0015562122],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017762813,0.00015051315,0.0046971356,0.00049528526,0.000034508987,0.00060405117,0.001891768,0.013943042,0.005457181,0.43260065,0.0100354,0.52991277],"study_design_scores_gemma":[0.00005875322,0.0006140698,0.0056968955,0.002004378,0.00008988414,0.0033776755,0.001005407,0.08628337,0.019122954,0.5725969,0.30896974,0.00017995654],"about_ca_topic_score_codex":0.0043087145,"about_ca_topic_score_gemma":0.0034120947,"teacher_disagreement_score":0.006153913,"about_ca_system_score_codex":0.0015310892,"about_ca_system_score_gemma":0.0018396692,"threshold_uncertainty_score":0.032545388},"labels":[],"label_agreement":null},{"id":"W4399169339","doi":"10.1109/wf-iot58464.2023.10539471","title":"A Software QA Framework for Autonomous Vehicle Open Source Application: OpenPilot","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Software deployment; Robustness (evolution); Software quality; Reliability engineering; Robustness testing; Regression testing; Software; Embedded system; Software engineering; Software system; Engineering; Software development; Operating system","score_opus":0.053631510355192924,"score_gpt":0.3356348568949485,"score_spread":0.28200334653975556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399169339","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007889695,0.00015625007,0.86543834,0.0002186753,0.0000469445,0.0003949741,0.0007017484,0.12126629,0.0038870862],"genre_scores_gemma":[0.26706588,0.00055224885,0.6857823,0.00047591393,0.0000719452,0.0012060953,0.0064314622,0.030116674,0.008297415],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983246,0.00044919862,0.00018243077,0.0002744629,0.0006341425,0.00013512919],"domain_scores_gemma":[0.9941526,0.0028733837,0.0005023661,0.0011723149,0.0010732362,0.00022602077],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004499593,0.0009243429,0.000540509,0.0017770412,0.0006589502,0.0019131283,0.0026874216,0.0012560477,0.008645927],"category_scores_gemma":[0.0117023755,0.0008665674,0.0010899673,0.0007332713,0.001223841,0.0033687057,0.0026103684,0.0021110615,0.0034000261],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011770258,0.0013193067,0.017669326,0.0024707909,0.00029517774,0.002530074,0.0039310446,0.121147186,0.05147312,0.13912761,0.09345676,0.56540257],"study_design_scores_gemma":[0.000291441,0.00061648275,0.0051319385,0.00062104303,0.0000782122,0.0015436613,0.00031187615,0.713113,0.034375932,0.054913655,0.18877463,0.00022807255],"about_ca_topic_score_codex":0.0031413126,"about_ca_topic_score_gemma":0.0027162717,"teacher_disagreement_score":0.008645927,"about_ca_system_score_codex":0.0008221125,"about_ca_system_score_gemma":0.0018151955,"threshold_uncertainty_score":0.028923512},"labels":[],"label_agreement":null},{"id":"W4399338452","doi":"10.11591/ijece.v14i4.pp4315-4324","title":"Optimized automated testing: test case generation and maintenance using latent semantic analysis-based TextRank and particle swarm optimization algorithms","year":2024,"lang":"en","type":"article","venue":"International Journal of Power Electronics and Drive Systems/International Journal of Electrical and Computer Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Computer science; Particle swarm optimization; Automatic summarization; Latent semantic analysis; Machine learning; Data mining; Software; Test case; Process (computing); Artificial intelligence; Multi-swarm optimization; Semantic analysis (machine learning); Algorithm; Programming language","score_opus":0.013636976004132873,"score_gpt":0.258746922159197,"score_spread":0.2451099461550641,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399338452","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042897027,0.00012949697,0.9523247,0.00016020109,0.000033590033,0.00022881055,0.000113229245,0.0027213143,0.0013916861],"genre_scores_gemma":[0.42678466,0.00009097425,0.5701968,0.0000908936,0.000036035737,0.00036790763,0.00060048647,0.00020948864,0.0016227353],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99891853,0.00039012716,0.00009587339,0.00019999173,0.00032308858,0.00007244015],"domain_scores_gemma":[0.9965783,0.0020414998,0.00042223273,0.00025860607,0.00062212907,0.00007725792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011560875,0.0010564377,0.0009139982,0.0016292972,0.0003711069,0.0009293093,0.0009870948,0.0008336449,0.0019160706],"category_scores_gemma":[0.005832809,0.0003410285,0.00067431293,0.0009389848,0.00039291655,0.0011817709,0.00052253227,0.0007115904,0.0005285738],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022841414,0.00044064902,0.0025305564,0.0001658623,0.00007332875,0.00015079998,0.00014226582,0.46703506,0.012063786,0.0032410247,0.00284713,0.5110812],"study_design_scores_gemma":[0.000015945341,0.00004490058,0.00024064683,0.0000034411519,0.0000061780324,0.000017938482,0.000011005261,0.9970898,0.0016121119,0.0007317836,0.00022230252,0.000003992453],"about_ca_topic_score_codex":0.0051658684,"about_ca_topic_score_gemma":0.0050379033,"teacher_disagreement_score":0.0051658684,"about_ca_system_score_codex":0.00080612395,"about_ca_system_score_gemma":0.0009597382,"threshold_uncertainty_score":0.010271609},"labels":[],"label_agreement":null},{"id":"W4399619591","doi":"10.1145/3672449","title":"An Empirical Study on the Characteristics of Database Access Bugs in Java Applications","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Database; SQL; Java; Database schema; Stored procedure; Query by Example; View; Commit; Database model; Database design; World Wide Web; Programming language; Web search query","score_opus":0.16223068351507128,"score_gpt":0.4138237030113616,"score_spread":0.2515930194962903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399619591","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9978023,0.0002369537,0.0008453349,0.000117253614,0.000006153577,0.000043296008,0.00032578866,0.000036783167,0.00058600865],"genre_scores_gemma":[0.9977277,0.00018102038,0.0011109817,0.000047532616,0.000009224236,0.00004319618,0.00064411253,0.000018976047,0.00021724912],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98892444,0.0026481408,0.0021577864,0.0014540509,0.004128757,0.0006868587],"domain_scores_gemma":[0.669917,0.1965958,0.09217562,0.008647342,0.029428454,0.0032358598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061616264,0.0003742577,0.0003194572,0.004793358,0.00062282174,0.0015716726,0.0009658005,0.0009133967,0.00094332354],"category_scores_gemma":[0.10643455,0.0004452152,0.0003747347,0.0045067132,0.0010266257,0.0030110432,0.0010011808,0.0011580465,0.0002466711],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000609809,0.00020502655,0.9826988,0.00016020046,0.00003600805,0.00024600865,0.0024348537,0.00024018611,0.000844613,0.000119679804,0.00030972823,0.012643878],"study_design_scores_gemma":[0.000005861335,0.0001850874,0.9918172,0.00007709398,0.000029603198,0.00070585456,0.0034573274,0.001976007,0.00074181677,0.00012452826,0.0008618966,0.000017785384],"about_ca_topic_score_codex":0.0036966866,"about_ca_topic_score_gemma":0.0069402256,"teacher_disagreement_score":0.0061616264,"about_ca_system_score_codex":0.0008276212,"about_ca_system_score_gemma":0.000984937,"threshold_uncertainty_score":0.032586217},"labels":[],"label_agreement":null},{"id":"W4399913729","doi":"10.1016/j.eswa.2024.124557","title":"Basis path coverage testing of MPI programs based on multi-task evolutionary optimization","year":2024,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"China Postdoctoral Science Foundation; National Natural Science Foundation of China","keywords":"Computer science; Task (project management); Path (computing); Basis (linear algebra); Machine learning; Artificial intelligence; Mathematics; Programming language","score_opus":0.028408197792053412,"score_gpt":0.26922609703886186,"score_spread":0.24081789924680844,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399913729","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6189334,0.00018427834,0.37212923,0.00037582844,0.000046484653,0.00008561724,0.00018141493,0.0028970542,0.0051668235],"genre_scores_gemma":[0.9345937,0.000033922868,0.06435997,0.00003851097,0.0000070401743,0.00005880819,0.00014430827,0.00018073544,0.0005831594],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99858,0.0006009619,0.000041000396,0.00012377246,0.00044853435,0.00020580002],"domain_scores_gemma":[0.9930656,0.0050537246,0.00032113885,0.0006007914,0.0008005733,0.00015808146],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012475656,0.00069641776,0.0007012953,0.0012061617,0.0006927846,0.0005470552,0.0013205581,0.0006913136,0.0018544401],"category_scores_gemma":[0.01056342,0.00026492248,0.0005528026,0.00071751646,0.00085561693,0.000926681,0.0009882075,0.0007327369,0.00013354373],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011776005,0.0003420323,0.01001905,0.00024916834,0.00010548169,0.0004022789,0.00022262067,0.82577085,0.022694036,0.021557568,0.00229084,0.11516842],"study_design_scores_gemma":[0.000022657945,0.000056840403,0.0005846111,0.000005269112,0.000009473454,0.000021787167,0.000015325255,0.99150825,0.0040968303,0.003549073,0.0001253814,0.0000044923545],"about_ca_topic_score_codex":0.0045542023,"about_ca_topic_score_gemma":0.004695305,"teacher_disagreement_score":0.0045542023,"about_ca_system_score_codex":0.000688076,"about_ca_system_score_gemma":0.001512854,"threshold_uncertainty_score":0.009055436},"labels":[],"label_agreement":null},{"id":"W4400242288","doi":"10.1145/3643991.3644914","title":"A Mutation-Guided Assessment of Acceleration Approaches for Continuous Integration: An Empirical Study of YourBase","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Waterloo","funders":"","keywords":"Acceleration; Computer science; Mutation; Empirical research; Physics; Mathematics; Genetics; Biology; Statistics","score_opus":0.18885613775120386,"score_gpt":0.421461520203657,"score_spread":0.23260538245245316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400242288","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9945589,0.00019037821,0.003707394,0.00009737347,0.00001027241,0.00018325333,0.00007328682,0.000086728425,0.0010924158],"genre_scores_gemma":[0.9870363,0.00009805231,0.011757853,0.00008647691,0.000007843784,0.00013910363,0.00022979653,0.000047898615,0.0005967197],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9836938,0.008099279,0.0009880519,0.0016633989,0.0051413556,0.00041406043],"domain_scores_gemma":[0.62945396,0.29760733,0.025876945,0.01986847,0.023672119,0.003521203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02554111,0.00063584116,0.00041835953,0.0018351384,0.0009006689,0.0019412636,0.0019599884,0.0014045279,0.0013468901],"category_scores_gemma":[0.16852114,0.0004046379,0.00056691945,0.0013092334,0.0016486947,0.0037880044,0.0016590514,0.0021614365,0.00046843072],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0051728804,0.017299261,0.5811397,0.0016456454,0.0007850002,0.0013133265,0.030127455,0.025746666,0.019965881,0.006936677,0.004565532,0.30530214],"study_design_scores_gemma":[0.0008368087,0.03131134,0.69572157,0.00079877325,0.00084483094,0.0020030204,0.018183945,0.20708664,0.017380444,0.006872679,0.01858356,0.00037636576],"about_ca_topic_score_codex":0.0028713236,"about_ca_topic_score_gemma":0.0040559727,"teacher_disagreement_score":0.02554111,"about_ca_system_score_codex":0.0015018085,"about_ca_system_score_gemma":0.0013614395,"threshold_uncertainty_score":0.13507593},"labels":[],"label_agreement":null},{"id":"W4400266862","doi":"10.1145/3643991.3649106","title":"Mining Our Way Back to Incremental Builds for DevOps Pipelines","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"DevOps; Computer science; Pipeline transport; Pipeline (software); Data science; Software engineering; Engineering; Software deployment; Programming language","score_opus":0.044570625776325916,"score_gpt":0.3248629375192195,"score_spread":0.2802923117428936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400266862","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021043116,0.0018378122,0.93212897,0.0060149366,0.00060727616,0.00038270582,0.0030807755,0.020456752,0.014447692],"genre_scores_gemma":[0.10693443,0.001437567,0.8625741,0.00108397,0.00018271421,0.0002208684,0.0100918235,0.0075547583,0.009919776],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98925483,0.0023389764,0.00077986973,0.0017657577,0.0051335273,0.0007270769],"domain_scores_gemma":[0.9531773,0.012597763,0.0024830622,0.020669607,0.010017749,0.0010545009],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069236425,0.0013665876,0.0008276863,0.0036991339,0.0020480908,0.006759992,0.004343778,0.001451084,0.007686127],"category_scores_gemma":[0.071434915,0.0020589228,0.0021637834,0.0033566724,0.0032300763,0.016563093,0.007315219,0.005610172,0.0046798233],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028359966,0.00027828827,0.017273668,0.0011395061,0.00020299619,0.0011791595,0.0040456154,0.018387726,0.009975804,0.17879012,0.09255774,0.6758857],"study_design_scores_gemma":[0.00006706874,0.00024062704,0.0051104166,0.0010041874,0.0001745918,0.0014155343,0.001704578,0.11659233,0.030295866,0.3101135,0.5330807,0.00020067059],"about_ca_topic_score_codex":0.0074906317,"about_ca_topic_score_gemma":0.015213341,"teacher_disagreement_score":0.007686127,"about_ca_system_score_codex":0.0013628062,"about_ca_system_score_gemma":0.00386741,"threshold_uncertainty_score":0.036616147},"labels":[],"label_agreement":null},{"id":"W4400484320","doi":"10.1145/3663529.3664459","title":"Enhancing Code Representation for Improved Graph Neural Network-Based Fault Localization","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Graph; Representation (politics); Artificial neural network; Code (set theory); Theoretical computer science; Artificial intelligence; Programming language","score_opus":0.027174466698568107,"score_gpt":0.3059534606372797,"score_spread":0.2787789939387116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400484320","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11774676,0.000517912,0.8652596,0.00036174327,0.00009385291,0.00007671434,0.00061581505,0.012316831,0.003010811],"genre_scores_gemma":[0.6899625,0.0002803269,0.30215287,0.00024791335,0.000034228233,0.00012301786,0.0024984325,0.00054346706,0.0041572736],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996356,0.000056691206,0.00001795587,0.000091424714,0.00014802563,0.000050350412],"domain_scores_gemma":[0.998906,0.00029819715,0.00016348973,0.00019716997,0.00038936216,0.00004574676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00032496956,0.0010679918,0.00047689144,0.0017592643,0.00027652542,0.0005908954,0.0012395722,0.00058134063,0.002035991],"category_scores_gemma":[0.0029570758,0.00026917667,0.00038486344,0.0010316672,0.00035758677,0.0013723752,0.00063954364,0.00076144916,0.0006785955],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018250919,0.00018732721,0.004805445,0.00017288626,0.000055894143,0.00015638302,0.00009315819,0.53834504,0.017843017,0.004788838,0.0076443865,0.42572507],"study_design_scores_gemma":[0.0000058215633,0.000029005041,0.00041879856,0.000005126265,0.000007802321,0.00002365208,0.000012631994,0.99273866,0.0038159366,0.0021883498,0.0007491833,0.0000050281205],"about_ca_topic_score_codex":0.011060492,"about_ca_topic_score_gemma":0.019839084,"teacher_disagreement_score":0.011060492,"about_ca_system_score_codex":0.0007862505,"about_ca_system_score_gemma":0.0011051918,"threshold_uncertainty_score":0.021992266},"labels":[],"label_agreement":null},{"id":"W4400484351","doi":"10.1145/3663529.3663804","title":"ATheNA-S: A Testing Tool for Simulink Models Driven by Software Requirements and Domain Expertise","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Software engineering; Software testing; Domain (mathematical analysis); Software; Programming language","score_opus":0.05609008688266483,"score_gpt":0.29812934737171304,"score_spread":0.24203926048904822,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400484351","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019017082,0.00015303527,0.8992399,0.00013915349,0.00004008167,0.0002865922,0.0009807852,0.07539316,0.004750239],"genre_scores_gemma":[0.24536406,0.00023553845,0.73688453,0.0001660338,0.000015911899,0.0007508075,0.003207708,0.007522951,0.0058524497],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99911016,0.0003108159,0.00008611124,0.00014195233,0.0002965275,0.0000543288],"domain_scores_gemma":[0.9964126,0.0025833044,0.00028200992,0.00039843383,0.0002554879,0.00006820476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016468704,0.0014378129,0.0005647471,0.0018148577,0.00033513474,0.0007690811,0.0017832248,0.0009864894,0.010830508],"category_scores_gemma":[0.0077461456,0.00078927836,0.00094761676,0.0004742323,0.0006676087,0.0013925981,0.0014224674,0.000989117,0.0015349786],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000837153,0.0007010105,0.011018082,0.0028045997,0.00039970363,0.0022357914,0.0015062678,0.35717273,0.095073,0.053815495,0.047159884,0.42727622],"study_design_scores_gemma":[0.00020533825,0.00026111695,0.0017144296,0.00028815435,0.00006513614,0.0009013712,0.000111728776,0.89885235,0.03075087,0.011506015,0.055256244,0.00008721933],"about_ca_topic_score_codex":0.0023161308,"about_ca_topic_score_gemma":0.004721503,"teacher_disagreement_score":0.010830508,"about_ca_system_score_codex":0.0005027166,"about_ca_system_score_gemma":0.0010385218,"threshold_uncertainty_score":0.036231697},"labels":[],"label_agreement":null},{"id":"W4400484638","doi":"10.1145/3663529.3663780","title":"AutoOffAB: Toward Automated Offline A/B Testing for Data-Driven Requirement Engineering","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; Kelowna General Hospital; University of British Columbia","funders":"","keywords":"Computer science; Software engineering; Reliability engineering; Engineering","score_opus":0.12448711876038791,"score_gpt":0.3383702706052155,"score_spread":0.2138831518448276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400484638","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016463554,0.00020385193,0.9073421,0.00029857695,0.00006478645,0.0005214431,0.00047199175,0.07156803,0.0030655812],"genre_scores_gemma":[0.18474893,0.00015779992,0.8059264,0.00036505505,0.00003094817,0.0004181862,0.0017532003,0.0042541954,0.0023453578],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9920874,0.0029522653,0.00046660664,0.0012135911,0.0028705818,0.00040962355],"domain_scores_gemma":[0.97527575,0.0105024185,0.0021248565,0.007847875,0.0033736783,0.00087550294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005741012,0.0020462496,0.00082172366,0.0030401654,0.00053221255,0.0027562394,0.0033771743,0.0012501094,0.004160813],"category_scores_gemma":[0.025987864,0.0012424262,0.0011923499,0.0008693206,0.001344142,0.0038682367,0.0038351787,0.0026341146,0.0033518856],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011053398,0.0012575717,0.020569734,0.0009413847,0.00021612979,0.00076328847,0.001555228,0.037952855,0.06458452,0.01818059,0.034739546,0.8181339],"study_design_scores_gemma":[0.000214271,0.00071957044,0.0065995483,0.00041180046,0.00007474453,0.0011568762,0.00055164355,0.8338977,0.072437435,0.034139868,0.049592935,0.00020361255],"about_ca_topic_score_codex":0.0036806713,"about_ca_topic_score_gemma":0.003930162,"teacher_disagreement_score":0.005741012,"about_ca_system_score_codex":0.00066398206,"about_ca_system_score_gemma":0.0020830927,"threshold_uncertainty_score":0.030361772},"labels":[],"label_agreement":null},{"id":"W4400485045","doi":"10.1145/3664646.3664772","title":"Chain of Targeted Verification Questions to Improve the Reliability of Code Generated by LLMs","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Reliability (semiconductor); Computer science; Code (set theory); Chain (unit); Reliability engineering; Risk analysis (engineering); Programming language; Business; Engineering","score_opus":0.012576388123733047,"score_gpt":0.26814752903102007,"score_spread":0.255571140907287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400485045","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11003991,0.0013049168,0.7816386,0.0010875188,0.00023411917,0.0012194195,0.0019729151,0.09980861,0.0026940028],"genre_scores_gemma":[0.3166335,0.00026464593,0.6647909,0.00087929855,0.0001275356,0.0006661804,0.0068363496,0.0049972157,0.004804309],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99131024,0.003263975,0.000681724,0.0022110734,0.0021899787,0.0003430201],"domain_scores_gemma":[0.9513153,0.03100107,0.004114128,0.006175181,0.00667866,0.00071558857],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054005724,0.002393904,0.0013966616,0.003126505,0.00063360744,0.0011341814,0.002744624,0.001986438,0.0056549483],"category_scores_gemma":[0.05278101,0.0007795744,0.0018366054,0.00087755476,0.000975633,0.0028089427,0.002971777,0.0017897527,0.0030499524],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001658953,0.0006639664,0.016836803,0.0024756028,0.00015509909,0.0011030257,0.002707157,0.018827315,0.06691499,0.0032902074,0.023976192,0.86139065],"study_design_scores_gemma":[0.00051352044,0.0012374859,0.011938366,0.00044828706,0.00029963927,0.0012866124,0.0009565541,0.8142296,0.11062907,0.01670956,0.041565105,0.00018623576],"about_ca_topic_score_codex":0.0030774889,"about_ca_topic_score_gemma":0.005268756,"teacher_disagreement_score":0.0056549483,"about_ca_system_score_codex":0.00094956625,"about_ca_system_score_gemma":0.0023524058,"threshold_uncertainty_score":0.028561294},"labels":[],"label_agreement":null},{"id":"W4400499041","doi":"10.1007/978-3-031-64171-8_18","title":"Pairing Security Advisories with Vulnerable Functions Using Open-Source LLMs","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Merck Canada Inc. (Canada)","funders":"","keywords":"Computer science; Pairing; Open source; Computer security; Programming language; Physics; Software","score_opus":0.026368676353523822,"score_gpt":0.26663719090995813,"score_spread":0.2402685145564343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400499041","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03927064,0.00024846077,0.8686582,0.0004842544,0.0002649476,0.0002807043,0.0003435216,0.057451684,0.03299763],"genre_scores_gemma":[0.58890474,0.00028777317,0.35683846,0.00035447275,0.00019033438,0.0003921477,0.0010441133,0.009570796,0.04241718],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9960681,0.0012744636,0.00032162695,0.00052025146,0.0013338838,0.0004815097],"domain_scores_gemma":[0.98850095,0.004130258,0.00071688613,0.0051543494,0.0010033188,0.000494216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003724151,0.0010141038,0.0008803688,0.0019371408,0.001155609,0.0031942893,0.001900949,0.0017184942,0.038846117],"category_scores_gemma":[0.019821258,0.0009829327,0.00086317223,0.0009339009,0.0014379716,0.0073447064,0.0061171455,0.0029816763,0.013264867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015414059,0.0005648826,0.0050151674,0.0004259844,0.00008032847,0.00084475084,0.001914901,0.012158148,0.030232925,0.26395616,0.04407615,0.6391892],"study_design_scores_gemma":[0.00017981292,0.00066504994,0.0022025444,0.0005675912,0.00013835217,0.0016493731,0.0006419886,0.22783634,0.15430865,0.32654,0.28502604,0.0002442528],"about_ca_topic_score_codex":0.0003417489,"about_ca_topic_score_gemma":0.00054486725,"teacher_disagreement_score":0.038846117,"about_ca_system_score_codex":0.0011467268,"about_ca_system_score_gemma":0.001144705,"threshold_uncertainty_score":0.12995315},"labels":[],"label_agreement":null},{"id":"W4400499064","doi":"10.1007/978-3-031-64171-8_5","title":"Modularizing Directed Greybox Fuzzing for Binaries over Multiple CPU Architectures","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Fuzz testing; Computer science; Programming language; Operating system; Parallel computing; Software","score_opus":0.022033722642859514,"score_gpt":0.26275811713292646,"score_spread":0.24072439449006694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400499064","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10054844,0.00037353742,0.8859154,0.00015936395,0.0001207107,0.00014012062,0.000074897805,0.008457253,0.004210341],"genre_scores_gemma":[0.6800153,0.0002516168,0.3131516,0.00021943895,0.000048426202,0.000096440264,0.00019784711,0.0011541856,0.0048650294],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989556,0.00018226098,0.000071239105,0.00022239331,0.00041151047,0.00015700208],"domain_scores_gemma":[0.9974988,0.0009790714,0.00019609954,0.0009500745,0.0003121659,0.000063807165],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011026043,0.0011695509,0.0006910639,0.0010401253,0.00043998382,0.0011403445,0.0017393007,0.0008164182,0.0035842855],"category_scores_gemma":[0.0038145918,0.0006886261,0.0010742326,0.00062095415,0.0013020892,0.0022574004,0.0020566648,0.0016400261,0.0007284602],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008165738,0.0002567713,0.00305113,0.0005610325,0.00021983245,0.00068132783,0.0005085915,0.2290272,0.13984266,0.102056354,0.0047461465,0.51823235],"study_design_scores_gemma":[0.000059914913,0.0002741036,0.0006890724,0.00013802058,0.00016908137,0.00031371563,0.00006269491,0.7210872,0.14720993,0.12323778,0.0066990657,0.0000593759],"about_ca_topic_score_codex":0.0014363957,"about_ca_topic_score_gemma":0.0026375705,"teacher_disagreement_score":0.0035842855,"about_ca_system_score_codex":0.0009439749,"about_ca_system_score_gemma":0.0008762639,"threshold_uncertainty_score":0.011990666},"labels":[],"label_agreement":null},{"id":"W4400582740","doi":"10.1145/3643759","title":"Understanding and Detecting Annotation-Induced Faults of Static Analyzers","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Annotation; Computer science; Natural language processing; Artificial intelligence","score_opus":0.05424952230509408,"score_gpt":0.25652039502094065,"score_spread":0.20227087271584657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400582740","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6221396,0.00056736596,0.36708477,0.00040961747,0.000045028988,0.00017859196,0.000329288,0.007297831,0.0019479914],"genre_scores_gemma":[0.89629686,0.00016887565,0.10223923,0.0000771854,0.000021167927,0.00007430744,0.0003603895,0.00034183203,0.00042016816],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9916937,0.00253628,0.00065454084,0.0013431265,0.0031185264,0.00065381185],"domain_scores_gemma":[0.95423377,0.02915878,0.006627103,0.004783086,0.0048486446,0.00034872806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045727636,0.0011541982,0.00066720543,0.004685577,0.00061719154,0.0016917605,0.0014184066,0.0014098623,0.00075236603],"category_scores_gemma":[0.034991015,0.0006691461,0.00075708795,0.0016579948,0.0014223083,0.0037904705,0.0013207167,0.0010525219,0.00019902937],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008111766,0.00076092355,0.33311468,0.0011323874,0.0002583624,0.004228123,0.012503698,0.04721941,0.13472216,0.018392196,0.0031012595,0.44375566],"study_design_scores_gemma":[0.00008216942,0.00067356176,0.11254516,0.00045735372,0.0005780004,0.0031794435,0.0034139138,0.67184573,0.17401804,0.022553435,0.010456982,0.00019615686],"about_ca_topic_score_codex":0.0037077777,"about_ca_topic_score_gemma":0.004932108,"teacher_disagreement_score":0.004685577,"about_ca_system_score_codex":0.001179266,"about_ca_system_score_gemma":0.0016514387,"threshold_uncertainty_score":0.024183393},"labels":[],"label_agreement":null},{"id":"W4400978763","doi":"10.1109/tse.2024.3433463","title":"Assessing Evaluation Metrics for Neural Test Oracle Generation","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Computer science; Oracle; Test (biology); Artificial neural network; Machine learning; Artificial intelligence; Data mining; Software engineering","score_opus":0.0715888736640746,"score_gpt":0.32531120179375506,"score_spread":0.25372232812968043,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400978763","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7561023,0.0070629446,0.21385626,0.0009375013,0.0002648613,0.0004984291,0.0020189257,0.012121408,0.0071372744],"genre_scores_gemma":[0.9271518,0.00041105808,0.06611537,0.00019817583,0.000051913004,0.00022887294,0.004573859,0.00031137688,0.0009575615],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9812439,0.00940529,0.0019100762,0.0022421458,0.0046207807,0.00057781604],"domain_scores_gemma":[0.9242682,0.05349312,0.0053389757,0.0059401887,0.009645438,0.0013141866],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0163487,0.0020596148,0.0011184674,0.00547723,0.00042894727,0.0017256219,0.0026717172,0.001831361,0.0010926625],"category_scores_gemma":[0.07321007,0.00048502474,0.00088132307,0.002708521,0.0009812658,0.0031634367,0.0015279104,0.0015608416,0.0004443592],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001117059,0.0006978048,0.05391304,0.00062249345,0.0007740631,0.0001518261,0.00019860655,0.58638835,0.00411688,0.002536953,0.006851782,0.34263113],"study_design_scores_gemma":[0.00004626533,0.0005470799,0.0045945193,0.0000372788,0.000055152526,0.0000606277,0.00004131041,0.98827296,0.004734738,0.0010869529,0.0004993793,0.000023735629],"about_ca_topic_score_codex":0.008149184,"about_ca_topic_score_gemma":0.009440368,"teacher_disagreement_score":0.0163487,"about_ca_system_score_codex":0.0040125526,"about_ca_system_score_gemma":0.0015552833,"threshold_uncertainty_score":0.086461246},"labels":[],"label_agreement":null},{"id":"W4401047823","doi":"10.1145/3680468","title":"Reinforcement Learning Informed Evolutionary Search for Autonomous Systems Testing","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Machine learning; Domain (mathematical analysis); Evolutionary algorithm; Population; Evolutionary robotics; Domain knowledge","score_opus":0.12685296961192286,"score_gpt":0.3405972594143513,"score_spread":0.21374428980242846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401047823","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09699502,0.00054953404,0.89646214,0.00040467695,0.00004569467,0.00011058748,0.000039608585,0.0009868313,0.004405802],"genre_scores_gemma":[0.8828777,0.00013492454,0.115334,0.00012891286,0.000016634593,0.00015879645,0.00006120846,0.00006734762,0.0012204107],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999171,0.00043877316,0.000030229214,0.0000804874,0.00020471937,0.00007472796],"domain_scores_gemma":[0.99744344,0.0018932886,0.0001888534,0.00015357393,0.0002379085,0.00008300968],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016370455,0.00071681075,0.0006920752,0.00066859374,0.00024220074,0.0004855109,0.0011193671,0.0008395492,0.0012917006],"category_scores_gemma":[0.0065676924,0.00035260696,0.0004245105,0.0003471333,0.0009840295,0.0006239649,0.000786807,0.0011269045,0.00017473234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021561085,0.000034111323,0.000538301,0.000021118101,0.000016288637,0.00003410996,0.000025314088,0.9785776,0.0007393286,0.0029572372,0.00016680582,0.01686815],"study_design_scores_gemma":[0.000007820371,0.000019071198,0.000061387225,0.00000340906,0.000002398907,0.000006319471,0.0000033798476,0.99793005,0.000213886,0.001605495,0.000145173,0.0000016169246],"about_ca_topic_score_codex":0.0030523369,"about_ca_topic_score_gemma":0.0025771346,"teacher_disagreement_score":0.0030523369,"about_ca_system_score_codex":0.00081350806,"about_ca_system_score_gemma":0.0010919076,"threshold_uncertainty_score":0.008657634},"labels":[],"label_agreement":null},{"id":"W4401073387","doi":"10.1109/tse.2024.3435067","title":"Mitigating the Uncertainty and Imprecision of Log-Based Code Coverage Without Requiring Additional Logging Statements","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Waterloo","funders":"","keywords":"Computer science; Logging; Code (set theory); Programming language; Data mining; Set (abstract data type)","score_opus":0.019196045384707247,"score_gpt":0.28167165167033514,"score_spread":0.2624756062856279,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401073387","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08253413,0.0006094046,0.9052518,0.0010726039,0.000061547624,0.00013611888,0.000599075,0.0067502274,0.0029850993],"genre_scores_gemma":[0.78267586,0.0003144323,0.21299684,0.0005004003,0.00009591638,0.0002378204,0.0011374921,0.0010523032,0.0009889533],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98171633,0.0044151624,0.0010729171,0.0028045133,0.009232014,0.00075904164],"domain_scores_gemma":[0.8632076,0.0845481,0.013448546,0.026939152,0.011051242,0.0008054895],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0095301615,0.0015122981,0.0013071501,0.0038493648,0.00082285,0.0034256966,0.0030548626,0.001357232,0.0009892347],"category_scores_gemma":[0.10093305,0.0012576196,0.0008183238,0.002349546,0.0021114082,0.007089182,0.00443108,0.0030085538,0.0004791465],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013267428,0.00046449498,0.07460446,0.0008760934,0.0004260296,0.00072428514,0.0023239972,0.30468017,0.0371257,0.03763637,0.007647926,0.5321637],"study_design_scores_gemma":[0.00005988428,0.0002501719,0.014634327,0.00023398381,0.00011861003,0.0004969799,0.00035153748,0.875133,0.046600565,0.0530657,0.008902496,0.00015284844],"about_ca_topic_score_codex":0.0045655393,"about_ca_topic_score_gemma":0.0054664663,"teacher_disagreement_score":0.0095301615,"about_ca_system_score_codex":0.0015105257,"about_ca_system_score_gemma":0.0025112163,"threshold_uncertainty_score":0.050400913},"labels":[],"label_agreement":null},{"id":"W4401172336","doi":"10.1007/978-3-031-47821-5_15","title":"Model Acceptance Testing","year":2024,"lang":"en","type":"book-chapter","venue":"CIGRE green books","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Opal-Rt Technologies (Canada)","funders":"","keywords":"Acceptance testing; Psychology; Computer science; Software engineering","score_opus":0.07991491499171106,"score_gpt":0.26715256573564944,"score_spread":0.1872376507439384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401172336","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002831596,0.004317975,0.16094321,0.0019739657,0.00077779574,0.00011135539,0.0003487373,0.0035001426,0.82519525],"genre_scores_gemma":[0.09292151,0.005697888,0.07281788,0.0018649026,0.00032575137,0.00027539695,0.0015477516,0.0026024312,0.8219464],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99848455,0.00028142508,0.00003501902,0.00014704315,0.0009698546,0.00008216669],"domain_scores_gemma":[0.998047,0.0008400819,0.00004901264,0.0004115477,0.0005976944,0.000054624335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009435239,0.0012402979,0.0008770834,0.0014056272,0.00085410674,0.0032685646,0.0016145317,0.0013632235,0.064749345],"category_scores_gemma":[0.004821496,0.00080688595,0.00067201076,0.0013382661,0.0010756975,0.003464161,0.0013235783,0.0025027904,0.026837738],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004142663,0.00014976818,0.00039354843,0.00017037106,0.000017554676,0.00013169662,0.0003485266,0.0032438624,0.0012192063,0.26659796,0.2264117,0.5012743],"study_design_scores_gemma":[0.000022124368,0.000088175075,0.00066279626,0.00036528215,0.000032913948,0.000533356,0.00027886633,0.02217769,0.0031774254,0.31797695,0.6546433,0.000041243984],"about_ca_topic_score_codex":0.002983294,"about_ca_topic_score_gemma":0.0048309295,"teacher_disagreement_score":0.064749345,"about_ca_system_score_codex":0.0010173726,"about_ca_system_score_gemma":0.001187484,"threshold_uncertainty_score":0.21660817},"labels":[],"label_agreement":null},{"id":"W4401420422","doi":"10.1145/3643656.3643899","title":"Predicting the Lifetime of Flaky Tests on Chrome","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Materials science; Reliability engineering; Computer science; Engineering","score_opus":0.016369370122478938,"score_gpt":0.2733729501834212,"score_spread":0.25700358006094226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401420422","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96933675,0.00087033835,0.022458654,0.0002326604,0.000038388545,0.000031481726,0.0020005908,0.0015407207,0.0034905204],"genre_scores_gemma":[0.9900664,0.00014548397,0.006690759,0.000040066512,0.000008146265,0.000012561317,0.0017488813,0.00013246996,0.0011552175],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992446,0.000086452,0.000041411542,0.00017744313,0.00031027096,0.00013976663],"domain_scores_gemma":[0.99158734,0.0051038936,0.0009080413,0.000471206,0.0014674644,0.00046201234],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013335268,0.00057814573,0.00036492894,0.0020832978,0.00027230696,0.0007333471,0.00067121803,0.00077942514,0.0019562645],"category_scores_gemma":[0.012071703,0.00026757488,0.0003689493,0.0008456421,0.0003895025,0.0011520414,0.00046044853,0.0007121507,0.00079157867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018894855,0.00033990684,0.38336104,0.00032554218,0.00015719554,0.0012879312,0.00036461707,0.42149615,0.035752982,0.0034160335,0.007917745,0.14369144],"study_design_scores_gemma":[0.000032491684,0.000354501,0.08154654,0.00006449276,0.00006399935,0.00045971165,0.00018437377,0.8850352,0.026214028,0.0032135388,0.0027674402,0.00006372377],"about_ca_topic_score_codex":0.008603476,"about_ca_topic_score_gemma":0.0127154505,"teacher_disagreement_score":0.008603476,"about_ca_system_score_codex":0.0007603441,"about_ca_system_score_gemma":0.0004087143,"threshold_uncertainty_score":0.017106831},"labels":[],"label_agreement":null},{"id":"W4401543513","doi":"10.1145/3643991.3645072","title":"Analyzing Developer Use of ChatGPT Generated Code in Open Source GitHub Projects","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Open source; Source code; Code (set theory); Programming language; Open source software; Code review; Software engineering; World Wide Web; Static program analysis; Software; Software development; Set (abstract data type)","score_opus":0.0982006071357781,"score_gpt":0.3196408188539973,"score_spread":0.2214402117182192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401543513","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9943949,0.000076992124,0.0031441257,0.00009878868,0.000009011334,0.00011278235,0.0003636576,0.0006552299,0.001144544],"genre_scores_gemma":[0.9848615,0.00012721858,0.010245094,0.00009655526,0.000011780206,0.00031270066,0.0019441142,0.0005936784,0.0018074743],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9922386,0.0036553447,0.00038764582,0.000928416,0.0024495858,0.00034055524],"domain_scores_gemma":[0.9011208,0.0705729,0.0104665095,0.0058203805,0.010270761,0.0017486889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005448634,0.00054494594,0.0002728358,0.003920659,0.0007417787,0.0014347128,0.0011209773,0.00085731683,0.0007034789],"category_scores_gemma":[0.076248705,0.0004694083,0.00025094717,0.0024738593,0.0012326209,0.0015604062,0.001933235,0.0010679251,0.00046080528],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011498242,0.0013839304,0.55775595,0.0015909667,0.00023763686,0.0071342653,0.12690233,0.008361461,0.028661232,0.0021943795,0.009587353,0.25504068],"study_design_scores_gemma":[0.00011320738,0.0015149551,0.8715589,0.0006980048,0.00013146884,0.0030009618,0.030862626,0.047459368,0.019239305,0.002100626,0.023035543,0.0002849689],"about_ca_topic_score_codex":0.0050880373,"about_ca_topic_score_gemma":0.012922414,"teacher_disagreement_score":0.005448634,"about_ca_system_score_codex":0.0010279039,"about_ca_system_score_gemma":0.0009798887,"threshold_uncertainty_score":0.028815508},"labels":[],"label_agreement":null},{"id":"W4401544398","doi":"10.1145/3643991.3644930","title":"Automating GUI-based Test Oracles for Mobile Apps","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"National Science Foundation","keywords":"Computer science; Mobile apps; Test (biology); World Wide Web","score_opus":0.01853026080516836,"score_gpt":0.30211460840373283,"score_spread":0.2835843475985645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401544398","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28734457,0.0015335436,0.5441214,0.0008480408,0.00019914635,0.002232205,0.004502652,0.15663332,0.0025852278],"genre_scores_gemma":[0.5452732,0.0004035375,0.4372232,0.0004520328,0.000078808895,0.0008710261,0.010532687,0.0036086987,0.0015568511],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98822355,0.003596469,0.0016674188,0.0023492833,0.0035047708,0.0006584691],"domain_scores_gemma":[0.9201079,0.03873017,0.01209873,0.015861887,0.011765989,0.0014353531],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0075808386,0.0022800984,0.00093673205,0.0057299943,0.0006719464,0.0022776297,0.002091002,0.0011821886,0.0013488156],"category_scores_gemma":[0.072103426,0.00087149354,0.0013970295,0.0017682592,0.0012198456,0.0032075942,0.0028162731,0.0019237815,0.0014002322],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070206955,0.0008447555,0.21307233,0.0016190676,0.00020744767,0.001773466,0.004648159,0.01788338,0.043200113,0.0034929714,0.019425467,0.69313073],"study_design_scores_gemma":[0.0003266146,0.0018187327,0.14033175,0.0010449613,0.0003919618,0.0045189043,0.0028818534,0.62601674,0.14758874,0.022029672,0.052463226,0.0005869253],"about_ca_topic_score_codex":0.0046110735,"about_ca_topic_score_gemma":0.007814669,"teacher_disagreement_score":0.0075808386,"about_ca_system_score_codex":0.0010072662,"about_ca_system_score_gemma":0.0021485619,"threshold_uncertainty_score":0.040091753},"labels":[],"label_agreement":null},{"id":"W4401632577","doi":"10.22215/etd/2024-16009","title":"Multilingual Fault Localization for Deep Learning Compilers","year":2024,"lang":"en","type":"dissertation","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Compiler; Codebase; Computer science; Deep learning; Programming language; Artificial intelligence; Software","score_opus":0.019550250925042846,"score_gpt":0.31594250928791834,"score_spread":0.29639225836287547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401632577","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26760986,0.0008163922,0.7010395,0.0004984688,0.0000951388,0.000121473,0.0007733974,0.02658988,0.0024559607],"genre_scores_gemma":[0.7227369,0.00016994239,0.2725312,0.00013102844,0.000017596436,0.000107696294,0.0015801446,0.00076591916,0.001959592],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986885,0.0002781618,0.00012872655,0.00033549953,0.00041906396,0.00015004845],"domain_scores_gemma":[0.9955225,0.001245718,0.0006342252,0.0008231911,0.0016125316,0.00016183566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001171971,0.0012935066,0.0005974991,0.0021055911,0.0005264211,0.0009789659,0.0015648953,0.0006421632,0.0018179268],"category_scores_gemma":[0.0066760904,0.0004346931,0.00091960456,0.0009075082,0.0007302455,0.0021608542,0.0016917919,0.0011850636,0.0006510197],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072070025,0.0004377609,0.021210985,0.0004490692,0.0001812916,0.00048494176,0.0004226213,0.3190277,0.04212764,0.010148078,0.009240469,0.5955488],"study_design_scores_gemma":[0.000024154815,0.00019787057,0.0020704751,0.000043186414,0.000060032664,0.00014631332,0.00013069929,0.93273693,0.049420573,0.011678913,0.0034607502,0.000030080395],"about_ca_topic_score_codex":0.006573015,"about_ca_topic_score_gemma":0.008697281,"teacher_disagreement_score":0.006573015,"about_ca_system_score_codex":0.0019442535,"about_ca_system_score_gemma":0.0021273254,"threshold_uncertainty_score":0.014106631},"labels":[],"label_agreement":null},{"id":"W4401724691","doi":"10.1145/3688842","title":"My Fuzzers Won’t Build: An Empirical Study of Fuzzing Build Failures","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Fuzz testing; Computer science; Software engineering; Software bug; Context (archaeology); Software; Set (abstract data type); Operating system; Programming language","score_opus":0.09234755339412065,"score_gpt":0.36974165303234796,"score_spread":0.2773940996382273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401724691","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99750334,0.00011669926,0.0011615547,0.00028992436,0.000004360011,0.000042680156,0.0002753023,0.000037787253,0.00056843983],"genre_scores_gemma":[0.9972646,0.00010504862,0.0015830352,0.00012333883,0.000008849613,0.00005238776,0.00059053034,0.00002726416,0.0002450115],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9938351,0.0024251395,0.0006752897,0.0008147748,0.0017778564,0.00047184378],"domain_scores_gemma":[0.76007736,0.1809156,0.031525474,0.010810126,0.013503847,0.0031676819],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.011479079,0.0005325733,0.0004165731,0.0033635243,0.0012856189,0.0014842574,0.0016814588,0.0015523524,0.0013397359],"category_scores_gemma":[0.09841094,0.0005941055,0.00046864024,0.0027228585,0.0019969721,0.0044907886,0.0014015264,0.002985418,0.00066038355],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029464392,0.001314647,0.947741,0.000187373,0.00010217747,0.0006523363,0.016649725,0.0031228059,0.0010516769,0.0014287434,0.003032471,0.02442245],"study_design_scores_gemma":[0.00006832298,0.0011390966,0.9128461,0.0003054225,0.00009765428,0.0017544753,0.026279146,0.046031892,0.0019692,0.0030164192,0.006371902,0.00012040791],"about_ca_topic_score_codex":0.008763247,"about_ca_topic_score_gemma":0.011497354,"teacher_disagreement_score":0.9885209,"about_ca_system_score_codex":0.0010457839,"about_ca_system_score_gemma":0.001102296,"threshold_uncertainty_score":0.060707927},"labels":[],"label_agreement":null},{"id":"W4401815668","doi":"10.1007/978-3-031-64136-7_5","title":"Verification and Validation of Quantum Software","year":2024,"lang":"en","type":"book-chapter","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Universität Innsbruck","keywords":"Computer science; Software verification; Software engineering; Verification and validation; Software; Programming language; Software construction; Software development; Mathematics; Statistics","score_opus":0.02981192027902278,"score_gpt":0.2559897412367009,"score_spread":0.2261778209576781,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401815668","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062748514,0.0011933483,0.86235917,0.0014224997,0.00030192608,0.00029666614,0.00037745782,0.009188935,0.062111486],"genre_scores_gemma":[0.41506374,0.0015719444,0.54664135,0.00051128946,0.00012205482,0.0003706424,0.0011585817,0.002394368,0.032166053],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99318075,0.0027702549,0.0003223778,0.00054816715,0.0029208758,0.00025760554],"domain_scores_gemma":[0.98234195,0.009727313,0.0005598523,0.003851634,0.003368505,0.00015071547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005706716,0.00048278066,0.00045441615,0.0012331659,0.0008090332,0.0023796707,0.0017501685,0.0010339593,0.008011532],"category_scores_gemma":[0.022159887,0.00037506816,0.00067107636,0.0010509994,0.0022542088,0.0028814536,0.0016718848,0.0014863684,0.002218162],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001877526,0.0002472876,0.0024787777,0.00090985646,0.000047401307,0.0005570162,0.0015083031,0.030152045,0.024937483,0.41398147,0.016345967,0.50864667],"study_design_scores_gemma":[0.00009782908,0.00036687727,0.0025200164,0.0013784742,0.000080805614,0.00089631043,0.0004063148,0.352809,0.16556156,0.22667786,0.24906047,0.00014449323],"about_ca_topic_score_codex":0.001898385,"about_ca_topic_score_gemma":0.001316186,"teacher_disagreement_score":0.008011532,"about_ca_system_score_codex":0.0011601731,"about_ca_system_score_gemma":0.0019369483,"threshold_uncertainty_score":0.030180335},"labels":[],"label_agreement":null},{"id":"W4401905737","doi":"10.1109/icst60714.2024.00052","title":"MOTIF: A tool for Mutation Testing with Fuzzing","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Université du Luxembourg; Entomological Society of America","keywords":"Fuzz testing; Computer science; Mutation testing; Software testing; Programming language; Mutation; Genetics; Biology; Software","score_opus":0.033953021041505084,"score_gpt":0.2770897851717789,"score_spread":0.24313676413027382,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401905737","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014344604,0.0005604545,0.8220597,0.00023918015,0.00010254568,0.00026934533,0.001711512,0.15651335,0.00419924],"genre_scores_gemma":[0.19099489,0.00049181597,0.786962,0.00037971028,0.00005925769,0.0008824879,0.0047129155,0.0109186545,0.0045982646],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979207,0.00051792443,0.0002187889,0.00036357678,0.00084449525,0.00013455536],"domain_scores_gemma":[0.99555886,0.0029816034,0.00038901088,0.00062349177,0.00035247853,0.00009449908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019750954,0.0017037604,0.00080324773,0.0027861055,0.00047571442,0.0012147051,0.0029827538,0.0018231701,0.009756657],"category_scores_gemma":[0.011295101,0.0010155075,0.0014029465,0.0010582332,0.0010830046,0.0021167477,0.0022765938,0.0016292809,0.0026951188],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013142729,0.0006028909,0.016773593,0.0024536366,0.0006169956,0.0024498545,0.0009270619,0.103320904,0.08245391,0.05452236,0.06942108,0.6651434],"study_design_scores_gemma":[0.0004324053,0.00056620524,0.004617913,0.00061842403,0.00021196218,0.004312212,0.00013581877,0.7420718,0.088331446,0.062100142,0.09635199,0.0002497844],"about_ca_topic_score_codex":0.001511625,"about_ca_topic_score_gemma":0.0015141583,"teacher_disagreement_score":0.009756657,"about_ca_system_score_codex":0.0005458562,"about_ca_system_score_gemma":0.0010928701,"threshold_uncertainty_score":0.032639265},"labels":[],"label_agreement":null},{"id":"W4401905742","doi":"10.1109/icst60714.2024.00055","title":"Insights into System Failures: ML-Assisted Testing and Failure Models for Cyber-Physical Systems","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Cyber-physical system; Computer science; Reliability engineering; System testing; Engineering; Software engineering; Operating system","score_opus":0.03310137089038518,"score_gpt":0.26533859974567836,"score_spread":0.2322372288552932,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401905742","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034697793,0.00042554908,0.9586154,0.0026145435,0.00003846832,0.00004960971,0.00014849051,0.00083565834,0.0025744264],"genre_scores_gemma":[0.77584165,0.0005898152,0.22082888,0.00036141812,0.00008247173,0.00012412274,0.00023262229,0.00016112032,0.001777933],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985032,0.00079659774,0.000066482295,0.0001465043,0.00039025513,0.00009688323],"domain_scores_gemma":[0.9839049,0.013037312,0.00090759667,0.001054395,0.000878252,0.00021751368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023555805,0.0009849314,0.00064079225,0.001336147,0.0002810283,0.0017370278,0.0015546817,0.0014603733,0.002176115],"category_scores_gemma":[0.020190628,0.00062335323,0.0009233077,0.00063939707,0.0019734558,0.0039433646,0.0012441362,0.0028821386,0.00030530203],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006877461,0.00011767447,0.0033378548,0.00009776773,0.000033484073,0.00016106312,0.00042800143,0.88412267,0.0015620834,0.062097147,0.0014222205,0.046551257],"study_design_scores_gemma":[0.0000052655364,0.000017033903,0.00022454507,0.000016343769,0.0000037291695,0.000018507344,0.000025600051,0.9705762,0.0004095909,0.028242836,0.0004540289,0.0000061766586],"about_ca_topic_score_codex":0.0039439737,"about_ca_topic_score_gemma":0.0041542985,"teacher_disagreement_score":0.0039439737,"about_ca_system_score_codex":0.0011244714,"about_ca_system_score_gemma":0.0010630889,"threshold_uncertainty_score":0.012457609},"labels":[],"label_agreement":null},{"id":"W4401907837","doi":"10.1109/icst60714.2024.00006","title":"Message from the Program Co-Chairs; ICST 2024","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Human–computer interaction","score_opus":0.024138685192223983,"score_gpt":0.31336838351121926,"score_spread":0.28922969831899525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401907837","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011599548,0.0024616325,0.0031849951,0.5691555,0.31312248,0.00036480735,0.0017583714,0.0014946654,0.10729758],"genre_scores_gemma":[0.007281129,0.0010883668,0.0016872421,0.12702776,0.035281498,0.00035456935,0.0009230883,0.00062873197,0.8257276],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985753,0.00017269683,0.000067276924,0.00025764288,0.00071623595,0.0002108495],"domain_scores_gemma":[0.9959209,0.00034700034,0.0001343271,0.00012920698,0.002141857,0.0013267183],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002210849,0.00070629653,0.00055601704,0.00066620373,0.0024892348,0.0033106143,0.0008429307,0.006910233,0.1364845],"category_scores_gemma":[0.0055574095,0.000285393,0.00069275126,0.0005629902,0.0005621414,0.0017097961,0.00141018,0.009060186,0.09959732],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001764209,0.000008585581,0.0000523589,0.000012181515,0.0000010368744,0.000027014868,0.0000067159704,0.000014361631,0.000085211774,0.00051134086,0.9954072,0.0038562473],"study_design_scores_gemma":[0.000009587981,0.000017036915,0.00023216197,0.000022192347,0.00000356588,0.000036930694,0.000041761446,0.000085070904,0.00023160824,0.0002734222,0.99903935,0.000007334078],"about_ca_topic_score_codex":0.005442609,"about_ca_topic_score_gemma":0.011082879,"teacher_disagreement_score":0.1364845,"about_ca_system_score_codex":0.0020485276,"about_ca_system_score_gemma":0.0042852303,"threshold_uncertainty_score":0.45658612},"labels":[],"label_agreement":null},{"id":"W4401908463","doi":"10.1109/icst60714.2024.00039","title":"Are We Testing or Being Tested? Exploring the Practical Applications of Large Language Models in Software Testing","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Software testing; Software reliability testing; Software performance testing; Software engineering; System integration testing; Test strategy; Software; Programming language; Software construction; Software system","score_opus":0.15819021582175738,"score_gpt":0.35758823244720933,"score_spread":0.19939801662545195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401908463","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7502638,0.0013855711,0.17564724,0.038099226,0.00006770314,0.0004436807,0.00015788602,0.0001976165,0.033737253],"genre_scores_gemma":[0.96723694,0.00033615436,0.030584402,0.0008959253,0.000012958243,0.0002560761,0.00004292013,0.000044324053,0.00059038185],"study_design_codex":"qualitative","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9574346,0.037088152,0.00069110177,0.0011054886,0.0028516718,0.00082905556],"domain_scores_gemma":[0.77046424,0.21188574,0.0061462526,0.0051762387,0.004878673,0.0014487855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.039219223,0.00043412842,0.00036307675,0.0017646822,0.0018868768,0.0067862994,0.0017450899,0.0019301203,0.0028716172],"category_scores_gemma":[0.11480318,0.0005807722,0.00043320138,0.0016694481,0.009812624,0.017470554,0.0036010493,0.0026600384,0.00038188175],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018168258,0.00053923775,0.06797178,0.0006355312,0.000046466335,0.0016521994,0.51824003,0.0030507331,0.0040530073,0.24249378,0.0030711233,0.15806447],"study_design_scores_gemma":[0.00009640924,0.00054867176,0.027813017,0.0017297256,0.00008588344,0.001770793,0.5723134,0.043443613,0.00331643,0.2852946,0.06342122,0.00016618687],"about_ca_topic_score_codex":0.0041386583,"about_ca_topic_score_gemma":0.0074424352,"teacher_disagreement_score":0.039219223,"about_ca_system_score_codex":0.004539677,"about_ca_system_score_gemma":0.0044623343,"threshold_uncertainty_score":0.2074135},"labels":[],"label_agreement":null},{"id":"W4402034738","doi":"10.32920/26871409","title":"Transformer Models for Automated Bug Triaging and Duplicate Bug Detection","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University; University of Manitoba; Systems, Applications & Products in Data Processing (Canada)","funders":"","keywords":"Computer science; Transformer; Engineering","score_opus":0.04005328244476049,"score_gpt":0.30055055257399427,"score_spread":0.2604972701292338,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402034738","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0506226,0.0011664697,0.9406954,0.0007372171,0.00010428109,0.00012803034,0.0005251612,0.0040290826,0.0019917737],"genre_scores_gemma":[0.779558,0.0010102078,0.20931634,0.0005147985,0.00017882399,0.00024214297,0.0025993937,0.0003985478,0.006181702],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99820495,0.0006539735,0.00013489748,0.0004963801,0.00036350696,0.000146206],"domain_scores_gemma":[0.9925368,0.0038963936,0.000768351,0.0010884937,0.0014699345,0.00024005726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004288184,0.001263007,0.0014048925,0.002579532,0.0005413116,0.0020666735,0.0022575154,0.001397597,0.0023631048],"category_scores_gemma":[0.0162392,0.0005744983,0.0016730138,0.002258695,0.0011139314,0.0033573348,0.0018986226,0.0025040992,0.0016018801],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005090376,0.0003758867,0.0064741042,0.00023849557,0.00020261791,0.00016508719,0.0002713634,0.6178409,0.0038820035,0.026313972,0.012052236,0.33167434],"study_design_scores_gemma":[0.000010464112,0.000040718583,0.00021809965,0.0000045573847,0.000011956877,0.000024338608,0.000011896809,0.988952,0.00047466587,0.009860239,0.0003833968,0.000007722746],"about_ca_topic_score_codex":0.009345104,"about_ca_topic_score_gemma":0.009172252,"teacher_disagreement_score":0.009345104,"about_ca_system_score_codex":0.001660664,"about_ca_system_score_gemma":0.0019021527,"threshold_uncertainty_score":0.022678316},"labels":[],"label_agreement":null},{"id":"W4402048137","doi":"10.1145/3690631","title":"T-Rec: Fine-Grained Language-Agnostic Program Reduction Guided by Lexical Syntax","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Syntax; Programming language; Reduction (mathematics); Natural language processing; Abstract syntax tree; Artificial intelligence","score_opus":0.0649629231789791,"score_gpt":0.3495486032377718,"score_spread":0.2845856800587927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402048137","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050681308,0.00077016297,0.84098715,0.00055515964,0.00022892162,0.000576186,0.0010305545,0.09902516,0.0061454065],"genre_scores_gemma":[0.26547408,0.00050951407,0.7040958,0.0009530962,0.00009859232,0.0008686104,0.0047839363,0.014503935,0.008712461],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972676,0.0005976339,0.00024979797,0.00058987667,0.00097317965,0.00032193292],"domain_scores_gemma":[0.99544275,0.0012029749,0.00041662186,0.0021046132,0.0007108597,0.00012226195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00153433,0.0019264423,0.0008665344,0.0016384668,0.00072732853,0.0015043512,0.0034991216,0.0011001214,0.0038635582],"category_scores_gemma":[0.0061787255,0.0007044721,0.002363468,0.0010793338,0.0021271778,0.0034243881,0.0030798921,0.0027019985,0.0020663117],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009564023,0.0008962805,0.011975479,0.0026294168,0.00043392184,0.0009013906,0.0012340929,0.048906077,0.16017139,0.08112842,0.060101997,0.6306651],"study_design_scores_gemma":[0.0005290227,0.0013969972,0.0048921686,0.00033536804,0.00060955225,0.0017888488,0.00056896766,0.48783326,0.25283775,0.0998551,0.14897357,0.0003793331],"about_ca_topic_score_codex":0.0032149984,"about_ca_topic_score_gemma":0.005898043,"teacher_disagreement_score":0.0038635582,"about_ca_system_score_codex":0.00088142516,"about_ca_system_score_gemma":0.0039102244,"threshold_uncertainty_score":0.01292485},"labels":[],"label_agreement":null},{"id":"W4402050968","doi":"10.1007/s10664-024-10477-1","title":"On combining commit grouping and build skip prediction to reduce redundant continuous integration activity","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Commit; Computer science; Database","score_opus":0.021112466439620347,"score_gpt":0.28530239328041973,"score_spread":0.2641899268407994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402050968","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1718079,0.0015056642,0.80555606,0.0012571798,0.0002403689,0.00027324577,0.0006288576,0.014174655,0.004556014],"genre_scores_gemma":[0.56350946,0.00034041377,0.4304113,0.00043854624,0.00012338156,0.00011979702,0.0014587725,0.0005792163,0.0030191133],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973877,0.00072152045,0.00015177742,0.0006156014,0.00084978685,0.00027353855],"domain_scores_gemma":[0.9879159,0.005293795,0.0007641556,0.0036806841,0.001984609,0.00036085243],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022525704,0.0014309543,0.0015332169,0.0024642586,0.000890379,0.0013739595,0.0026095342,0.0012701324,0.0028831777],"category_scores_gemma":[0.012861779,0.0006032662,0.000674581,0.0023650615,0.00069336087,0.0036562039,0.002183112,0.0017686051,0.0010219339],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006268266,0.0010995874,0.018312886,0.0001777127,0.00013680117,0.000117930766,0.00021639712,0.06797091,0.01682017,0.0030859609,0.009734232,0.8817005],"study_design_scores_gemma":[0.00009877834,0.00038937965,0.006435947,0.000051818253,0.00015137455,0.00013259174,0.0001725056,0.969964,0.010675444,0.009599287,0.0022880277,0.000040778297],"about_ca_topic_score_codex":0.008296199,"about_ca_topic_score_gemma":0.020324666,"teacher_disagreement_score":0.008296199,"about_ca_system_score_codex":0.00052424043,"about_ca_system_score_gemma":0.0028789435,"threshold_uncertainty_score":0.016495824},"labels":[],"label_agreement":null},{"id":"W4402264853","doi":"10.62973/12-105","title":"OGC OWS-9 - OWS Context evaluation IP Engineering Report","year":2013,"lang":"en","type":"report","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Defence Science and Technology Group; Natural Resources Canada; U.S. Army Corps of Engineers; Defence Science and Technology Laboratory; National Geospatial-Intelligence Agency; Federal Aviation Administration; U.S. Geological Survey; National Aeronautics and Space Administration","keywords":"Context (archaeology); Computer science; Geography; Archaeology","score_opus":0.07937924010193935,"score_gpt":0.3261484479895635,"score_spread":0.24676920788762413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402264853","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08417916,0.0006674369,0.23775542,0.00859318,0.002501385,0.013547585,0.027209196,0.028504832,0.59704185],"genre_scores_gemma":[0.21195324,0.0012571529,0.37114218,0.0039515896,0.0007817089,0.007204399,0.098657124,0.019824237,0.2852285],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.97388965,0.0045088185,0.0007360448,0.00090307574,0.01788679,0.0020756354],"domain_scores_gemma":[0.9672824,0.0031566338,0.0006844271,0.004383731,0.022906518,0.0015861952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020863196,0.001234579,0.00086344866,0.0029860171,0.0021630935,0.006414826,0.0025841887,0.0018513729,0.033191558],"category_scores_gemma":[0.03306141,0.00073999807,0.0008683355,0.001825977,0.0011724221,0.005672406,0.0035124563,0.0034040017,0.01942329],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013180672,0.0035057587,0.005473931,0.00052765216,0.00007367064,0.0004836927,0.0011550763,0.011017948,0.024148555,0.04115505,0.5438706,0.36726996],"study_design_scores_gemma":[0.00051590876,0.0017648193,0.006023459,0.00031369552,0.00008623846,0.00030722088,0.0010158435,0.03317755,0.07883579,0.006648186,0.8711302,0.0001810668],"about_ca_topic_score_codex":0.03473437,"about_ca_topic_score_gemma":0.014945547,"teacher_disagreement_score":0.03473437,"about_ca_system_score_codex":0.004815524,"about_ca_system_score_gemma":0.011217254,"threshold_uncertainty_score":0.11103684},"labels":[],"label_agreement":null},{"id":"W4402352815","doi":"10.1109/ijcnn60899.2024.10650368","title":"Knowledge-Informed Auto-Penetration Testing Based on Reinforcement Learning with Reward Machine","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada; Concordia University","keywords":"Reinforcement learning; Computer science; Penetration (warfare); Machine learning; Artificial intelligence; Reinforcement; Psychology; Engineering; Social psychology; Operations research","score_opus":0.03125521645134659,"score_gpt":0.28878440336496997,"score_spread":0.25752918691362336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402352815","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.072027184,0.00019375014,0.9231459,0.00033632832,0.000030373907,0.000094854986,0.00005758867,0.0021717567,0.0019423615],"genre_scores_gemma":[0.9418698,0.000052807092,0.05697531,0.00012185444,0.000010630801,0.00009625131,0.000063604704,0.00007099361,0.00073870545],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986344,0.0005014436,0.000066021676,0.00030588938,0.00026840233,0.0002237864],"domain_scores_gemma":[0.99447423,0.0036182129,0.00062634976,0.00047166235,0.0005531296,0.00025645364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018080515,0.0011149904,0.0013100654,0.00058027694,0.00041943556,0.00072796404,0.0018400047,0.0011618659,0.001811661],"category_scores_gemma":[0.009015256,0.0005813401,0.0005916883,0.00037950394,0.0013814508,0.0015446866,0.0015663883,0.0016658676,0.0002641464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008585272,0.000075718315,0.0014736138,0.000047195925,0.000023890307,0.00009883774,0.000054011816,0.94666874,0.0013962323,0.0044783265,0.0005104218,0.045087196],"study_design_scores_gemma":[0.000007455676,0.000025601808,0.00006747443,0.0000031870884,0.000003242784,0.000010133434,0.0000029653831,0.99783933,0.00045492302,0.0015044172,0.00007816519,0.000002980398],"about_ca_topic_score_codex":0.005396568,"about_ca_topic_score_gemma":0.0040503563,"teacher_disagreement_score":0.005396568,"about_ca_system_score_codex":0.0010373658,"about_ca_system_score_gemma":0.0021819482,"threshold_uncertainty_score":0.010730326},"labels":[],"label_agreement":null},{"id":"W4402442339","doi":"10.1145/3650212.3652126","title":"LPR: Large Language Models-Aided Program Reduction","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Government of Canada; University of Waterloo","funders":"","keywords":"Computer science; Compiler; Programming language; JavaScript; Generality; Semantics (computer science); Debugging; Reduction (mathematics)","score_opus":0.02354237607111619,"score_gpt":0.3102948756575512,"score_spread":0.286752499586435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402442339","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010640606,0.0006702538,0.900268,0.0004475357,0.00008045067,0.0002264819,0.0007740781,0.083386466,0.0035060379],"genre_scores_gemma":[0.13905935,0.0004544553,0.8388128,0.0006636634,0.00006658679,0.0005935909,0.004484278,0.010237559,0.005627699],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968207,0.0008801534,0.00016939484,0.000659784,0.0012325988,0.0002372492],"domain_scores_gemma":[0.9967615,0.0013648366,0.00025483526,0.0011121299,0.00044753018,0.00005923298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001575269,0.0026380585,0.0009989276,0.0015147872,0.00071979564,0.0014225561,0.004061808,0.0010115742,0.0071047656],"category_scores_gemma":[0.005562772,0.0011375686,0.0036843524,0.0010405638,0.0015395324,0.0036975576,0.0031869856,0.0033454944,0.002904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000569888,0.00038097732,0.0039945547,0.0017197372,0.00034184917,0.0006003331,0.0008009172,0.22930324,0.05215862,0.051021334,0.05959605,0.5995125],"study_design_scores_gemma":[0.000133216,0.00018257972,0.0004940412,0.00007829363,0.00013406968,0.00020760087,0.00013470778,0.8831634,0.037335053,0.040374022,0.037694644,0.00006850169],"about_ca_topic_score_codex":0.0057118107,"about_ca_topic_score_gemma":0.0139050735,"teacher_disagreement_score":0.0071047656,"about_ca_system_score_codex":0.0013403448,"about_ca_system_score_gemma":0.0035940993,"threshold_uncertainty_score":0.02376777},"labels":[],"label_agreement":null},{"id":"W4402442387","doi":"10.1145/3650212.3680359","title":"ThinkRepair: Self-Directed Automated Program Repair","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science","score_opus":0.012641153618666493,"score_gpt":0.29300937128207494,"score_spread":0.28036821766340847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402442387","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015158274,0.0012292363,0.18604046,0.00053078355,0.00023753077,0.0003909406,0.0061994484,0.78848135,0.0017320081],"genre_scores_gemma":[0.13537014,0.00072971336,0.7703008,0.0019959882,0.00014762976,0.0007974615,0.064367145,0.018312218,0.007978952],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99589795,0.0010333486,0.00027280167,0.0015876344,0.0010065989,0.00020154654],"domain_scores_gemma":[0.9917514,0.0036104475,0.00060581334,0.0028738808,0.00086422137,0.00029430218],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026645013,0.0040525342,0.0014156218,0.0026264675,0.0006667817,0.0014345946,0.007179865,0.0026924028,0.0057167737],"category_scores_gemma":[0.013345205,0.0014355609,0.003057632,0.0014354229,0.0012364857,0.0039383,0.0038707142,0.0034788854,0.006549443],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013996981,0.0009435675,0.0093181,0.0025515016,0.00041979106,0.00068453565,0.0009475555,0.031507332,0.022985125,0.004130649,0.32519802,0.59991395],"study_design_scores_gemma":[0.0008838839,0.0007568881,0.0037751407,0.00018918018,0.0002470671,0.00095256313,0.00040080468,0.8319394,0.037341736,0.015499973,0.1078175,0.00019587722],"about_ca_topic_score_codex":0.007994171,"about_ca_topic_score_gemma":0.016775766,"teacher_disagreement_score":0.007994171,"about_ca_system_score_codex":0.0010311165,"about_ca_system_score_gemma":0.0023298208,"threshold_uncertainty_score":0.019124508},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"not_applicable","genre":"software","about_ca_system":false,"about_ca_topic":false,"confidence":"medium"}],"label_agreement":"split"},{"id":"W4402442473","doi":"10.1145/3650212.3680332","title":"Semantic Constraint Inference for Web Form Test Generation","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Inference; Constraint (computer-aided design); Test (biology); Natural language processing; Artificial intelligence; Semantic Web; Information retrieval; Programming language; Mathematics; Geology","score_opus":0.046113461862748825,"score_gpt":0.30886525854924146,"score_spread":0.2627517966864926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402442473","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018413158,0.00015734948,0.958282,0.0006430744,0.00005336124,0.0002757861,0.0014155533,0.01716557,0.003594179],"genre_scores_gemma":[0.30294773,0.00017871232,0.6879247,0.0005234727,0.00004220715,0.00030018212,0.004342515,0.0019705328,0.0017699017],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949326,0.0022512209,0.00027468108,0.00066985324,0.0016148624,0.00025678604],"domain_scores_gemma":[0.9878163,0.008533084,0.00058528926,0.001478405,0.0014090441,0.00017794583],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029930482,0.0013657777,0.00068296614,0.0022595022,0.0006532948,0.0016214888,0.0024164852,0.0011296683,0.006786807],"category_scores_gemma":[0.02355283,0.000601497,0.0018612922,0.0012967131,0.0019641782,0.0031782507,0.0024438181,0.0018962177,0.0013158865],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004951877,0.00038778852,0.008283139,0.0010228477,0.00017550329,0.0010515015,0.0005521163,0.38051903,0.015177626,0.069708735,0.026736,0.49589053],"study_design_scores_gemma":[0.000052989162,0.000042175758,0.00037107713,0.000057600326,0.000028923987,0.00017424209,0.00008020637,0.9399783,0.012871501,0.038788002,0.00753277,0.000022319242],"about_ca_topic_score_codex":0.008224616,"about_ca_topic_score_gemma":0.017683012,"teacher_disagreement_score":0.008224616,"about_ca_system_score_codex":0.0018170088,"about_ca_system_score_gemma":0.0033245175,"threshold_uncertainty_score":0.022704065},"labels":[],"label_agreement":null},{"id":"W4402457261","doi":"10.1145/3650212.3680354","title":"Domain Adaptation for Code Model-Based Unit Test Case Generation","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; York University","funders":"","keywords":"Computer science; Unit testing; Code generation; Adaptation (eye); Code (set theory); Test (biology); Domain (mathematical analysis); Code coverage; Programming language; Operating system; Software; Key (lock)","score_opus":0.0997005380221137,"score_gpt":0.32023556997296465,"score_spread":0.22053503195085095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402457261","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18803018,0.0015455579,0.76186043,0.0005497088,0.00016459588,0.00040668348,0.002128442,0.03945634,0.005858027],"genre_scores_gemma":[0.69253975,0.00037220502,0.2941643,0.00057794835,0.000034490156,0.00045564995,0.0077571636,0.0014859986,0.002612527],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99851793,0.00053489103,0.00011509914,0.0004159104,0.0002911282,0.00012501521],"domain_scores_gemma":[0.99587554,0.0021088433,0.0003068138,0.00088140054,0.0007096667,0.00011777869],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001396034,0.0011765256,0.0005382195,0.0012531327,0.00018329614,0.0006574562,0.0017495404,0.00076352,0.0024170212],"category_scores_gemma":[0.008358624,0.00039110126,0.0009975515,0.0009925696,0.00052922365,0.0014351712,0.0010986782,0.001671959,0.0011754144],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003431004,0.00050813664,0.012027055,0.0004500298,0.00014849298,0.00048668918,0.00026501567,0.34081885,0.032210343,0.0038079747,0.016096136,0.5928382],"study_design_scores_gemma":[0.0000581355,0.000135332,0.0012018073,0.000028877503,0.00003667732,0.00018498098,0.000054116823,0.9712673,0.017872415,0.004266193,0.00487153,0.000022649767],"about_ca_topic_score_codex":0.0033504122,"about_ca_topic_score_gemma":0.004838611,"teacher_disagreement_score":0.0033504122,"about_ca_system_score_codex":0.0008882517,"about_ca_system_score_gemma":0.0012108237,"threshold_uncertainty_score":0.008085787},"labels":[],"label_agreement":null},{"id":"W4402457710","doi":"10.1145/3650212.3680398","title":"Characterizing and Detecting Program Representation Faults of Static Analysis Frameworks","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Representation (politics); Static analysis; Program analysis; Programming language","score_opus":0.02245917158366384,"score_gpt":0.333704196811921,"score_spread":0.31124502522825714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402457710","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9235874,0.00073666155,0.06982581,0.00034603832,0.000033919376,0.00021480108,0.00037071618,0.003821309,0.0010634414],"genre_scores_gemma":[0.94843954,0.00015062315,0.05041567,0.00007339057,0.000009662876,0.00008202628,0.0004219527,0.00016720757,0.00023992051],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98467505,0.0045864186,0.0012273401,0.002397525,0.0060254512,0.0010881613],"domain_scores_gemma":[0.90492433,0.053639486,0.019162335,0.011269569,0.01020083,0.0008034187],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00826075,0.00096212036,0.0005622775,0.0058497246,0.0007662409,0.0012550767,0.0016174004,0.0011770227,0.0006611355],"category_scores_gemma":[0.0726452,0.0006137479,0.00079160655,0.0023621137,0.0014482278,0.003208921,0.0018411856,0.0011694389,0.00015366082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007353036,0.000761462,0.48462844,0.0011903781,0.00030257975,0.001654871,0.008147421,0.019896345,0.055686373,0.007994234,0.0030006364,0.41600192],"study_design_scores_gemma":[0.00016849114,0.0022701789,0.3415979,0.0011024977,0.0010652116,0.005068516,0.0070664226,0.46950024,0.13537586,0.016113413,0.02027997,0.00039126523],"about_ca_topic_score_codex":0.0031509823,"about_ca_topic_score_gemma":0.0046391743,"teacher_disagreement_score":0.00826075,"about_ca_system_score_codex":0.001214679,"about_ca_system_score_gemma":0.002547909,"threshold_uncertainty_score":0.043687582},"labels":[],"label_agreement":null},{"id":"W4402473702","doi":"10.1109/ccece59415.2024.10667115","title":"Optimized Test Data Generation for Path Testing Using Improved Combined Fitness Function with Modified Particle Swarm Optimization Algorithm","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Particle swarm optimization; Fitness function; Algorithm; Computer science; Path (computing); Multi-swarm optimization; Test functions for optimization; Test data; Mathematical optimization; Function (biology); Mathematics; Machine learning; Genetic algorithm","score_opus":0.10710881190519414,"score_gpt":0.2976885004140749,"score_spread":0.19057968850888077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402473702","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056850936,0.00025972817,0.93953824,0.00012096672,0.000047778405,0.000085374355,0.000046127523,0.0008164814,0.0022344051],"genre_scores_gemma":[0.60359496,0.00014391792,0.3938697,0.000056670757,0.000021344133,0.00021954703,0.00018809608,0.000106870895,0.001798854],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99959594,0.00009624264,0.000028023169,0.00006547212,0.00016849728,0.00004588912],"domain_scores_gemma":[0.99938715,0.00028881402,0.000059184506,0.00005218414,0.00018787428,0.000024754661],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00056472933,0.000641296,0.00058582885,0.0008911637,0.0002677681,0.00053093745,0.00090375805,0.0007269027,0.0010016804],"category_scores_gemma":[0.0018532394,0.00023283575,0.0006087727,0.00060473644,0.00026699592,0.000538397,0.00042617088,0.00049233815,0.00018387134],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008636901,0.00008079559,0.0016779652,0.00006253076,0.000045511104,0.00014476452,0.000064670625,0.85535157,0.008975807,0.0025691774,0.0008884418,0.13005243],"study_design_scores_gemma":[0.000011862595,0.00003479084,0.00022666623,0.000002434073,0.0000073992846,0.00002627002,0.000003258669,0.99762803,0.001404806,0.0002921835,0.0003580672,0.0000042404613],"about_ca_topic_score_codex":0.005069889,"about_ca_topic_score_gemma":0.003124486,"teacher_disagreement_score":0.005069889,"about_ca_system_score_codex":0.000473415,"about_ca_system_score_gemma":0.0010326481,"threshold_uncertainty_score":0.010080755},"labels":[],"label_agreement":null},{"id":"W4402516053","doi":"10.1145/3695989","title":"Non-Flaky and Nearly Optimal Time-Based Treatment of Asynchronous Wait Web Tests","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Computer science; Asynchronous communication; Web application; World Wide Web; Computer network","score_opus":0.04587187782534556,"score_gpt":0.3016080723484287,"score_spread":0.2557361945230831,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402516053","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7590974,0.0017751998,0.22248016,0.0009440966,0.00018294896,0.00025796547,0.0036932803,0.00799569,0.0035731664],"genre_scores_gemma":[0.9498145,0.00014399066,0.044590868,0.00017941714,0.00009049286,0.0001879933,0.003395208,0.00052613346,0.0010714607],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9917149,0.0019089959,0.0009827043,0.002407829,0.0021404242,0.0008452082],"domain_scores_gemma":[0.9581916,0.022816142,0.0061232555,0.0074166195,0.004152961,0.0012993781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004589241,0.0008567235,0.0008538467,0.0026489557,0.0007763374,0.0018704757,0.0018115428,0.001466494,0.001800057],"category_scores_gemma":[0.050830305,0.0004366964,0.00090329914,0.0019588338,0.0010650404,0.0021139183,0.0013442124,0.0015780524,0.00061208685],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003311094,0.000986592,0.2859802,0.0011563099,0.00034197408,0.0018613106,0.002250596,0.19166388,0.04629816,0.01696142,0.020213595,0.42897484],"study_design_scores_gemma":[0.00024522634,0.0008158573,0.077810444,0.00012273664,0.00018334614,0.0013091922,0.00086850126,0.8513567,0.022857375,0.0319124,0.0123982215,0.00012006645],"about_ca_topic_score_codex":0.0032956146,"about_ca_topic_score_gemma":0.0043923073,"teacher_disagreement_score":0.004589241,"about_ca_system_score_codex":0.0011106811,"about_ca_system_score_gemma":0.0021523547,"threshold_uncertainty_score":0.024270535},"labels":[],"label_agreement":null},{"id":"W4402526955","doi":"10.1145/3679006.3685068","title":"Using Category Partition to Detect Metamorphic Relations","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Partition (number theory); Metamorphic rock; Computer science; Artificial intelligence; Geology; Mathematics; Combinatorics; Petrology","score_opus":0.08057092530792093,"score_gpt":0.32251630077330123,"score_spread":0.2419453754653803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402526955","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2251607,0.00022678947,0.7655771,0.0001862071,0.000031304244,0.0001792349,0.00028393246,0.001290956,0.0070637637],"genre_scores_gemma":[0.84052384,0.00007140721,0.15746829,0.000074290496,0.00001953895,0.00016812084,0.00047124445,0.00013673474,0.0010664995],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961916,0.00088990654,0.00027289728,0.0009111187,0.001317553,0.0004169039],"domain_scores_gemma":[0.98966485,0.004942678,0.0011190871,0.0017845548,0.0020410947,0.00044763013],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002514498,0.00052394293,0.00049854704,0.004937439,0.0011070289,0.0019845162,0.0011105187,0.0013051647,0.0024078886],"category_scores_gemma":[0.011825936,0.00040167128,0.00068187877,0.0017768296,0.0033685332,0.0033581867,0.0035024397,0.0012037427,0.00027290187],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009640496,0.00018053035,0.078822635,0.0003975175,0.00015731786,0.0013474956,0.0051400824,0.027433086,0.053276077,0.5040742,0.00245653,0.32575053],"study_design_scores_gemma":[0.000070384864,0.00065730186,0.03428185,0.00023083283,0.00016318217,0.0034160702,0.0022230244,0.33134907,0.058133483,0.5487673,0.02049793,0.00020951536],"about_ca_topic_score_codex":0.0034454623,"about_ca_topic_score_gemma":0.0019034831,"teacher_disagreement_score":0.004937439,"about_ca_system_score_codex":0.0009808022,"about_ca_system_score_gemma":0.00085670885,"threshold_uncertainty_score":0.013298094},"labels":[],"label_agreement":null},{"id":"W4402526960","doi":"10.1145/3678722.3685532","title":"Directed or Undirected: Investigating Fuzzing Strategies in a CI/CD Setup (Registered Report)","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Naval Information Warfare Center Pacific; Advanced Research Projects Agency; Defense Advanced Research Projects Agency","keywords":"Fuzz testing; Computer science; Programming language; Software","score_opus":0.0665116022280347,"score_gpt":0.3284333257113033,"score_spread":0.2619217234832686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402526960","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90136683,0.000848134,0.07166734,0.00058665324,0.000136955,0.0011998742,0.0022626594,0.011324873,0.010606625],"genre_scores_gemma":[0.8683488,0.0002510466,0.123393014,0.0003357657,0.000018572133,0.0006004734,0.0035770787,0.00095230836,0.0025229848],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958045,0.0014095802,0.00029519372,0.0009555041,0.0012023171,0.0003329591],"domain_scores_gemma":[0.96048963,0.025428433,0.0016480673,0.007881904,0.0037122383,0.00083973305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053188307,0.0011329914,0.00042804313,0.0014297604,0.00067231024,0.0013414435,0.0027117352,0.0012632537,0.0036153416],"category_scores_gemma":[0.04296864,0.00059647916,0.00060552085,0.0011155943,0.0011497582,0.0025757,0.001826994,0.0018898058,0.0006158436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005430632,0.0069687245,0.088365994,0.004321674,0.00065434934,0.0009039614,0.0034593777,0.17991708,0.086638644,0.021766491,0.042127587,0.5594455],"study_design_scores_gemma":[0.0015143406,0.00699691,0.053881567,0.00040355636,0.0005361017,0.0009096644,0.0022189678,0.7731624,0.11370439,0.018683743,0.027688166,0.00030014457],"about_ca_topic_score_codex":0.007325056,"about_ca_topic_score_gemma":0.009839593,"teacher_disagreement_score":0.007325056,"about_ca_system_score_codex":0.0014609625,"about_ca_system_score_gemma":0.0014765004,"threshold_uncertainty_score":0.028129041},"labels":[],"label_agreement":null},{"id":"W4402527174","doi":"10.1145/3678722.3685529","title":"The Havoc Paradox in Generator-Based Fuzzing (Registered Report)","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Fuzz testing; Generator (circuit theory); Computer science; Computer security; Programming language; Physics; Power (physics); Software","score_opus":0.036559975035849394,"score_gpt":0.29856286036877877,"score_spread":0.26200288533292937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402527174","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44448838,0.00037843236,0.54142094,0.00073405396,0.00008518143,0.0001751694,0.00021683429,0.0045234864,0.007977558],"genre_scores_gemma":[0.8996997,0.000084796186,0.09848532,0.00015976667,0.0000109026605,0.000060218452,0.00018594816,0.00034992746,0.000963476],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9923928,0.0027053484,0.00037755174,0.000854231,0.0032921091,0.00037798175],"domain_scores_gemma":[0.97043186,0.019981746,0.0016153823,0.00563805,0.0020213334,0.00031157758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070928778,0.00046697093,0.00053991925,0.0011118863,0.0005383344,0.001984739,0.0013357812,0.0009375687,0.0014976235],"category_scores_gemma":[0.039080225,0.00046750915,0.00054894417,0.000715885,0.002263665,0.002949505,0.0015905874,0.0014599278,0.00020380128],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001992599,0.000585249,0.037489172,0.0009333781,0.0002994632,0.0015790155,0.0030860477,0.16269314,0.10625544,0.2963075,0.007165656,0.38161325],"study_design_scores_gemma":[0.00014436174,0.0010086434,0.007846788,0.00013726272,0.00014981169,0.001600212,0.00031393644,0.68440616,0.195638,0.09870846,0.009901809,0.00014453297],"about_ca_topic_score_codex":0.0013268712,"about_ca_topic_score_gemma":0.0014743914,"teacher_disagreement_score":0.0070928778,"about_ca_system_score_codex":0.00085849105,"about_ca_system_score_gemma":0.0009877321,"threshold_uncertainty_score":0.03751117},"labels":[],"label_agreement":null},{"id":"W4402527462","doi":"10.1145/3678720.3685320","title":"Abstract Debugging with GobPie","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Debugging; Computer science; Programming language; Software engineering","score_opus":0.01472623548697537,"score_gpt":0.2515278555332744,"score_spread":0.236801620046299,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402527462","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008656677,0.00016755515,0.91124153,0.00028229115,0.00009720403,0.00011460076,0.000483069,0.06987928,0.009077844],"genre_scores_gemma":[0.18627112,0.00044024867,0.7850157,0.000676206,0.00006911266,0.00025399018,0.0020917694,0.014235783,0.01094607],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971788,0.0008526523,0.00017635903,0.00042984137,0.0011247109,0.0002376683],"domain_scores_gemma":[0.9931671,0.0033551562,0.00051439594,0.0015266387,0.0012261139,0.00021055923],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029643322,0.0012697593,0.000810852,0.0017787818,0.00050003454,0.0025843163,0.0018260628,0.001159977,0.011750667],"category_scores_gemma":[0.012960098,0.001028257,0.00063609146,0.0007735061,0.0010234402,0.0049182894,0.002989024,0.0027809672,0.004003488],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028303275,0.00039143217,0.00785608,0.0013391111,0.00014353191,0.0014829303,0.0032223586,0.028347928,0.07382673,0.1628639,0.072050124,0.64564556],"study_design_scores_gemma":[0.00059306424,0.00068285887,0.003582182,0.0011595467,0.000220404,0.0024184873,0.0005309189,0.31129593,0.15861566,0.12690116,0.39365926,0.00034055044],"about_ca_topic_score_codex":0.001118346,"about_ca_topic_score_gemma":0.0011499318,"teacher_disagreement_score":0.011750667,"about_ca_system_score_codex":0.000497396,"about_ca_system_score_gemma":0.0012207888,"threshold_uncertainty_score":0.03930992},"labels":[],"label_agreement":null},{"id":"W4402571068","doi":"10.1109/icstw60967.2024.00031","title":"Replay-Based Continual Learning for Test Case Prioritization","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Prioritization; Computer science; Test (biology); Software engineering; Artificial intelligence; Machine learning; Process management; Engineering","score_opus":0.01782511303937027,"score_gpt":0.2913492331547891,"score_spread":0.2735241201154188,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402571068","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.081719786,0.0011469275,0.90550125,0.0006206718,0.00010609282,0.0004942878,0.00036045472,0.008087899,0.0019626443],"genre_scores_gemma":[0.73913664,0.00024714012,0.256207,0.00050739193,0.0001048035,0.00080706883,0.0012176328,0.00032866225,0.0014436777],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.994319,0.0019262391,0.0005283233,0.0016185158,0.0012361434,0.0003717671],"domain_scores_gemma":[0.96855295,0.022967245,0.0021043161,0.0024125008,0.0032137013,0.0007493666],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00838257,0.0025595594,0.0021496294,0.0034771895,0.00083223445,0.0013767462,0.0038462947,0.0014469816,0.003141305],"category_scores_gemma":[0.035376918,0.0012311196,0.0013902707,0.001783858,0.0015309586,0.0028025564,0.0025171211,0.004014003,0.000792322],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005890634,0.0007125208,0.009819996,0.00030967747,0.0002096584,0.00017981601,0.00036327713,0.562035,0.003157984,0.0037254426,0.0030325481,0.41586506],"study_design_scores_gemma":[0.000033098535,0.000081456375,0.0003563247,0.0000121910625,0.000016208936,0.000019625893,0.000017434852,0.9958046,0.0006698201,0.0026874659,0.0002918445,0.000009931019],"about_ca_topic_score_codex":0.009449108,"about_ca_topic_score_gemma":0.01030994,"teacher_disagreement_score":0.009449108,"about_ca_system_score_codex":0.0020453036,"about_ca_system_score_gemma":0.0031142859,"threshold_uncertainty_score":0.04433179},"labels":[],"label_agreement":null},{"id":"W4402571187","doi":"10.1109/icstw60967.2024.00014","title":"An End-to-End Test Case Prioritization Framework using Optimized Machine Learning Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"End-to-end principle; Computer science; Test (biology); Prioritization; End user; End milling; End of history; Artificial intelligence; Engineering; Operating system; Management science; Mechanical engineering","score_opus":0.04802987512244553,"score_gpt":0.3161299016249306,"score_spread":0.26810002650248504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402571187","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009463906,0.00009951234,0.95747936,0.00016963625,0.000021173988,0.00021466948,0.00017796486,0.031457007,0.0009167847],"genre_scores_gemma":[0.3024613,0.000087991124,0.69231087,0.00021478215,0.000041613046,0.00043928315,0.0010689342,0.0021082957,0.0012670304],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9943863,0.0019821469,0.00041446826,0.00096218765,0.0019023665,0.000352582],"domain_scores_gemma":[0.98641914,0.006662546,0.0017168975,0.0018859173,0.0028333294,0.0004821304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007142549,0.0022563052,0.0010268986,0.0029128098,0.0005038146,0.0024968667,0.0029927576,0.0011665362,0.0028633305],"category_scores_gemma":[0.024065752,0.0008676118,0.0009555557,0.0010014414,0.0007897094,0.0024024474,0.0020295726,0.0025598113,0.001154484],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052725466,0.0007626419,0.009676156,0.00024094021,0.00017564214,0.00035359187,0.00030142267,0.5228529,0.014916685,0.009341763,0.011668513,0.42918256],"study_design_scores_gemma":[0.00001921833,0.000068505105,0.0004904525,0.000017589657,0.00001501652,0.000046425295,0.00001772388,0.9904046,0.0046364153,0.0032170685,0.0010503456,0.000016620694],"about_ca_topic_score_codex":0.0069468706,"about_ca_topic_score_gemma":0.009291032,"teacher_disagreement_score":0.007142549,"about_ca_system_score_codex":0.0017155691,"about_ca_system_score_gemma":0.0034066315,"threshold_uncertainty_score":0.037773848},"labels":[],"label_agreement":null},{"id":"W4402571211","doi":"10.1109/icstw60967.2024.00062","title":"TCPGraphix: A Visualization Tool for ML-Powered Test Case Prioritization Data Analysis","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Visualization; Prioritization; Data visualization; Test (biology); Data mining; Engineering; Process management; Geology","score_opus":0.04666377528138762,"score_gpt":0.3563540356949651,"score_spread":0.3096902604135775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402571211","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013849017,0.00050561834,0.71369725,0.0010917993,0.00023209311,0.00059270783,0.0138040595,0.249162,0.007065467],"genre_scores_gemma":[0.17722438,0.0009980836,0.7615891,0.0007330744,0.00021178806,0.0021011496,0.021034114,0.030282639,0.005825539],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972433,0.00084533566,0.00032371623,0.00034372075,0.0010916976,0.00015221957],"domain_scores_gemma":[0.9664352,0.025757387,0.001553081,0.0032160536,0.0025247117,0.00051361666],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064280718,0.0022599988,0.0009118805,0.007736058,0.0006859246,0.004578704,0.0025382144,0.001226334,0.022548819],"category_scores_gemma":[0.033981092,0.0008940277,0.0011095871,0.0039482038,0.0007889911,0.004020144,0.0036687162,0.0029274721,0.0031612224],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001301535,0.0005259969,0.012950187,0.0032831477,0.00040962765,0.0015949974,0.006546546,0.05042353,0.016662914,0.047383863,0.25003132,0.6088863],"study_design_scores_gemma":[0.00058312784,0.000549039,0.010735709,0.0015990996,0.00025204942,0.0017685579,0.0013182103,0.5331144,0.055194054,0.086172976,0.30827972,0.0004330815],"about_ca_topic_score_codex":0.0033797754,"about_ca_topic_score_gemma":0.0030515492,"teacher_disagreement_score":0.022548819,"about_ca_system_score_codex":0.0011488255,"about_ca_system_score_gemma":0.0024318045,"threshold_uncertainty_score":0.075433314},"labels":[],"label_agreement":null},{"id":"W4402571307","doi":"10.1109/icstw60967.2024.00021","title":"Annotating Control-Flow Graphs for Formalized Test Coverage Criteria","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Queen's University","funders":"","keywords":"Computer science; Control flow; Test (biology); Control flow graph; Control (management); Programming language; Software engineering; Artificial intelligence","score_opus":0.017486802881724738,"score_gpt":0.2954803116486411,"score_spread":0.27799350876691636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402571307","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006785348,0.00005936694,0.98781955,0.00016122517,0.000027072007,0.0002375919,0.0004952783,0.003398944,0.0010156196],"genre_scores_gemma":[0.08915808,0.00013088249,0.90560985,0.00018569433,0.000031104573,0.00048112095,0.0019343797,0.0016413061,0.00082763605],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98919046,0.0043793404,0.0012005814,0.0014737085,0.0029807317,0.00077525707],"domain_scores_gemma":[0.95294935,0.03420071,0.0031776442,0.0041562202,0.0050897924,0.0004264017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00829849,0.0023793264,0.0012010764,0.0067119584,0.0013384873,0.0040178937,0.002082382,0.0024430372,0.004487192],"category_scores_gemma":[0.04824405,0.0012091559,0.0031290327,0.0033462641,0.003304545,0.005102865,0.0037297176,0.0032205794,0.0010027956],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037069607,0.00035958266,0.011912177,0.0015021204,0.00017560403,0.0017565043,0.0031495388,0.39434665,0.026273794,0.2698803,0.008595984,0.2816771],"study_design_scores_gemma":[0.00006758388,0.000079584686,0.00094337115,0.00044273905,0.00008029873,0.00033458445,0.0003149041,0.6934144,0.024508646,0.25870204,0.021007283,0.00010460172],"about_ca_topic_score_codex":0.008849689,"about_ca_topic_score_gemma":0.010464182,"teacher_disagreement_score":0.008849689,"about_ca_system_score_codex":0.0034067933,"about_ca_system_score_gemma":0.003455769,"threshold_uncertainty_score":0.04388714},"labels":[],"label_agreement":null},{"id":"W4402860132","doi":"10.1145/3697014","title":"ZigZagFuzz: Interleaved Fuzzing of Program Options and Files","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nexen (Canada)","funders":"Samsung; Hanyang University","keywords":"Fuzz testing; Computer science; Programming language; Software","score_opus":0.08854984899731846,"score_gpt":0.34345537315099234,"score_spread":0.25490552415367385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402860132","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26438805,0.0009776702,0.70005983,0.00032568295,0.0000812728,0.00038461722,0.0005641338,0.030590376,0.0026283367],"genre_scores_gemma":[0.68955743,0.00017512668,0.30538964,0.00038133867,0.00001783679,0.00019902145,0.0009875728,0.0009192528,0.0023727408],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998381,0.00029576643,0.00013766583,0.0005141038,0.0005061843,0.00016521537],"domain_scores_gemma":[0.99599457,0.0018807907,0.00037908327,0.0013020472,0.00033518136,0.00010833979],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014136719,0.0013071863,0.0006002002,0.0013288921,0.00042025844,0.0008907681,0.0022658221,0.0009941766,0.0019075369],"category_scores_gemma":[0.0060335933,0.00057669147,0.0010449721,0.0005789679,0.0014645681,0.0021382775,0.001315637,0.0010559473,0.00037847494],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024681375,0.0003349831,0.022966437,0.0007492111,0.00032298654,0.0009766446,0.0007739832,0.10534372,0.2134185,0.016409757,0.0056126066,0.63062304],"study_design_scores_gemma":[0.00027035215,0.0009978644,0.006815648,0.00012714109,0.0002972605,0.001186538,0.00018272578,0.68393433,0.2715059,0.022285482,0.012203938,0.00019278459],"about_ca_topic_score_codex":0.004794141,"about_ca_topic_score_gemma":0.0058460324,"teacher_disagreement_score":0.004794141,"about_ca_system_score_codex":0.00082518207,"about_ca_system_score_gemma":0.0010557193,"threshold_uncertainty_score":0.009532511},"labels":[],"label_agreement":null},{"id":"W4402979393","doi":"10.1109/tse.2024.3469582","title":"LTM: Scalable and Black-Box Similarity-Based Test Suite Minimization Based on Language Models","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trent University; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada; Science Foundation Ireland","keywords":"Computer science; Test suite; Suite; Scalability; Black box; Minification; Similarity (geometry); Test (biology); Software testing; Programming language; Artificial intelligence; Natural language processing; Test case; Software; Machine learning; Operating system","score_opus":0.012067979201598519,"score_gpt":0.22770173116824316,"score_spread":0.21563375196664464,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402979393","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031785652,0.0003180002,0.9539198,0.00018508993,0.00003783467,0.00022951531,0.0003411955,0.012057973,0.001124961],"genre_scores_gemma":[0.39652836,0.00017620507,0.5952758,0.00037782153,0.00005999994,0.0006499056,0.003016325,0.0013652277,0.002550332],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966059,0.0008353331,0.00023753673,0.00066258304,0.0013909796,0.00026775317],"domain_scores_gemma":[0.99552,0.0019641798,0.00065144454,0.00083419535,0.0008231046,0.00020703071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016065646,0.001679135,0.0014687987,0.0019453132,0.0005230513,0.0012289311,0.0030413985,0.0011825558,0.0022722355],"category_scores_gemma":[0.010397906,0.00057719706,0.0016731913,0.0011800139,0.0009957781,0.0023583544,0.0026070967,0.0017391035,0.0010189726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052190194,0.0004163557,0.004876078,0.0004906138,0.00022026907,0.00041257232,0.00023204515,0.3393781,0.032141805,0.0124070505,0.008228276,0.6006749],"study_design_scores_gemma":[0.00003225071,0.00016019013,0.0003633985,0.000012245695,0.000021235279,0.00010990687,0.000030004861,0.9821755,0.009828903,0.0062555247,0.0009962813,0.000014587585],"about_ca_topic_score_codex":0.0050331466,"about_ca_topic_score_gemma":0.006445027,"teacher_disagreement_score":0.0050331466,"about_ca_system_score_codex":0.0015425693,"about_ca_system_score_gemma":0.0028987718,"threshold_uncertainty_score":0.011192143},"labels":[],"label_agreement":null},{"id":"W4403060990","doi":"10.1109/tse.2024.3472476","title":"FlakyFix: Using Large Language Models for Predicting Flaky Test Fix Categories and Test Code Repair","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Ottawa","funders":"Science Foundation Ireland","keywords":"Computer science; Test (biology); Programming language; Code (set theory); Code coverage; Software engineering; Reliability engineering; Software; Engineering","score_opus":0.017848771159971646,"score_gpt":0.2560316308607052,"score_spread":0.23818285970073355,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403060990","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40254417,0.0022846542,0.48961008,0.0017724764,0.00037093167,0.00056620984,0.013530003,0.08664207,0.0026794255],"genre_scores_gemma":[0.7101169,0.00030427313,0.25890473,0.00059677137,0.00007660829,0.00052267156,0.025082335,0.001397629,0.0029979805],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986584,0.00039223622,0.00008742434,0.000527372,0.00022336705,0.000111216024],"domain_scores_gemma":[0.992694,0.004812276,0.000622818,0.00078034523,0.00077725394,0.00031326726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002000266,0.0022184532,0.0008043775,0.0023329507,0.000510273,0.0011345375,0.0019239719,0.0019511944,0.0019694094],"category_scores_gemma":[0.010637897,0.00055427593,0.0014841435,0.00088575174,0.00062636845,0.0021545608,0.0013298234,0.0031485015,0.0015094983],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012101313,0.0012451935,0.06855758,0.00087334437,0.0004568301,0.0010673669,0.0007733957,0.37742865,0.019940326,0.002850067,0.037881806,0.48771533],"study_design_scores_gemma":[0.000044093453,0.00015821887,0.003011846,0.000032931664,0.000040091567,0.00009059631,0.00006794455,0.9862278,0.0054253256,0.0027822193,0.002078792,0.00004023271],"about_ca_topic_score_codex":0.016934121,"about_ca_topic_score_gemma":0.030209368,"teacher_disagreement_score":0.016934121,"about_ca_system_score_codex":0.0013902454,"about_ca_system_score_gemma":0.0017391375,"threshold_uncertainty_score":0.03367114},"labels":[],"label_agreement":null},{"id":"W4403064409","doi":"10.3389/frobt.2024.1346580","title":"AAT4IRS: automated acceptance testing for industrial robotic systems","year":2024,"lang":"en","type":"article","venue":"Frontiers in Robotics and AI","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; École de Technologie Supérieure; Université du Québec à Chicoutimi","funders":"","keywords":"Computer science; Software engineering; Process (computing); Software; Acceptance testing; Robustness testing; Industrial robot; Robot; Robustness (evolution); Artificial intelligence; Operating system","score_opus":0.04334930426113736,"score_gpt":0.28442739993703703,"score_spread":0.24107809567589966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403064409","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11763341,0.0003481675,0.85162556,0.0003578375,0.00007233471,0.00044048572,0.00031303818,0.025232978,0.0039761923],"genre_scores_gemma":[0.5146417,0.00019253294,0.48033142,0.00023218858,0.000036230595,0.0003851291,0.0009014893,0.0009863259,0.002292924],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9926093,0.0031004662,0.00043685446,0.00064269104,0.0029066168,0.00030397598],"domain_scores_gemma":[0.98745626,0.0077474765,0.0012989393,0.001100891,0.0021527736,0.00024365858],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035280725,0.0010873134,0.00044519064,0.0011776659,0.0003987407,0.0009373802,0.0023130786,0.0011263089,0.0029922526],"category_scores_gemma":[0.012414591,0.00038012507,0.0009684339,0.00048794574,0.0012414396,0.001645365,0.00091661717,0.0010768556,0.0008068356],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011738307,0.0013504031,0.014824142,0.0013254526,0.00022861321,0.0025906602,0.0017538107,0.11570955,0.26994234,0.023763014,0.012177113,0.55516106],"study_design_scores_gemma":[0.0003174257,0.0023895549,0.0053951894,0.00017694598,0.00011510846,0.0024963773,0.00022025674,0.76696473,0.18521018,0.012382598,0.024176424,0.0001551375],"about_ca_topic_score_codex":0.0018771259,"about_ca_topic_score_gemma":0.0014460931,"teacher_disagreement_score":0.0035280725,"about_ca_system_score_codex":0.00058731064,"about_ca_system_score_gemma":0.0011198203,"threshold_uncertainty_score":0.01865846},"labels":[],"label_agreement":null},{"id":"W4403427009","doi":"10.5753/sblp.2024.3471","title":"Semantic conflict detection via dynamic analysis","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fundação de Amparo à Ciência e Tecnologia do Estado de Pernambuco; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Coordenação de Aperfeiçoamento de Pessoal de Nível Superior","keywords":"Computer science; Artificial intelligence; Information retrieval; Natural language processing","score_opus":0.011761843050324437,"score_gpt":0.274519036368925,"score_spread":0.26275719331860053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403427009","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10638785,0.00090129796,0.8661084,0.0005934596,0.0001187705,0.0003640281,0.0014646903,0.016148798,0.007912655],"genre_scores_gemma":[0.6274065,0.00034888746,0.36512098,0.0003014378,0.000061850165,0.00040080334,0.0031873856,0.0015004864,0.0016716425],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98123735,0.0044295746,0.0012620867,0.0033283504,0.008600805,0.0011418494],"domain_scores_gemma":[0.97042555,0.015139866,0.0041797445,0.0039987927,0.005769365,0.00048662422],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056754756,0.0019072127,0.0011150116,0.010111474,0.0012690575,0.0037684683,0.002496571,0.001607573,0.0025228565],"category_scores_gemma":[0.034938496,0.0008755591,0.0014671813,0.004082812,0.0016344025,0.0048582563,0.0052594976,0.0020574487,0.00096640526],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010815974,0.00063659856,0.075999334,0.0013723936,0.0005919678,0.003186547,0.0041377237,0.063657284,0.075069286,0.043952134,0.012330752,0.71798444],"study_design_scores_gemma":[0.00007923055,0.00024052456,0.013044722,0.00033272794,0.0003542266,0.0024124188,0.0014622296,0.7800102,0.0973518,0.06871833,0.03575471,0.00023893199],"about_ca_topic_score_codex":0.0033227263,"about_ca_topic_score_gemma":0.0031810289,"teacher_disagreement_score":0.010111474,"about_ca_system_score_codex":0.0013936092,"about_ca_system_score_gemma":0.0025375348,"threshold_uncertainty_score":0.03001517},"labels":[],"label_agreement":null},{"id":"W4403536012","doi":"10.1145/3691620.3695551","title":"Efficient Incremental Code Coverage Analysis for Regression Test Suites","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Regression testing; Code coverage; Programming language; Code (set theory); Test (biology); Regression analysis; Parallel computing; Machine learning; Software; Software development","score_opus":0.02530939162931477,"score_gpt":0.3040384391541011,"score_spread":0.27872904752478633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403536012","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.082559764,0.0007609864,0.8986915,0.00024850748,0.00004839276,0.0003501848,0.00062783505,0.013730001,0.0029829105],"genre_scores_gemma":[0.5836295,0.00031437114,0.40813544,0.00018786249,0.000093611314,0.0005242425,0.0036399795,0.0015023453,0.0019726923],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99492294,0.0013283245,0.00026520825,0.0006489964,0.0023172633,0.0005172686],"domain_scores_gemma":[0.9807027,0.012913818,0.001420915,0.0016960574,0.002931999,0.0003345404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002157812,0.0015290225,0.0011522087,0.005210008,0.0005233707,0.0013778001,0.0020582587,0.0007478181,0.0031768354],"category_scores_gemma":[0.024497662,0.000584402,0.001545208,0.001913396,0.0008700863,0.0018061429,0.0016426117,0.0013347504,0.0010857163],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006774716,0.00037316186,0.020734644,0.0007231883,0.0002231228,0.00074235274,0.00031642182,0.25163847,0.06401573,0.014256256,0.0074662734,0.6388329],"study_design_scores_gemma":[0.00006068381,0.00018898795,0.0032507582,0.0000571693,0.00007407181,0.00026323897,0.00005008547,0.9615894,0.020005226,0.012016456,0.002417079,0.000026910204],"about_ca_topic_score_codex":0.0040644184,"about_ca_topic_score_gemma":0.0047521275,"teacher_disagreement_score":0.005210008,"about_ca_system_score_codex":0.0010729925,"about_ca_system_score_gemma":0.0017730439,"threshold_uncertainty_score":0.0114117265},"labels":[],"label_agreement":null},{"id":"W4403536208","doi":"10.1145/3691620.3695533","title":"Developer-Applied Accelerations in Continuous Integration: A Detection Approach and Catalog of Patterns","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Huawei Technologies (Canada); University of Waterloo","funders":"","keywords":"Computer science; Database","score_opus":0.022935272753777453,"score_gpt":0.2514001286771509,"score_spread":0.22846485592337343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403536208","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19109836,0.0018836593,0.78467035,0.0014685547,0.00013934268,0.0011947415,0.0018793655,0.0071974387,0.010468128],"genre_scores_gemma":[0.37245715,0.0009192169,0.615162,0.00032941755,0.00006995964,0.00074828934,0.002361217,0.00075416296,0.0071985247],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98617536,0.0019430501,0.0015358182,0.0032789237,0.006421266,0.00064555614],"domain_scores_gemma":[0.94199854,0.019410776,0.009356849,0.013105621,0.01440273,0.0017254726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007188317,0.0013744732,0.001121091,0.011004844,0.002009398,0.004415493,0.0030178046,0.0020457162,0.0019561064],"category_scores_gemma":[0.04525841,0.0013690152,0.0013523464,0.008591084,0.002660227,0.0076978547,0.004628415,0.0026143193,0.0010408424],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005755208,0.0007436763,0.22855227,0.0011923135,0.00019181932,0.002776452,0.010570497,0.007582504,0.022812244,0.05054393,0.009870096,0.6645887],"study_design_scores_gemma":[0.00024098142,0.001189891,0.12741949,0.0016302292,0.00065904943,0.012768091,0.0086455075,0.5164892,0.059673604,0.1653431,0.105284326,0.00065653207],"about_ca_topic_score_codex":0.009121434,"about_ca_topic_score_gemma":0.011630692,"teacher_disagreement_score":0.011004844,"about_ca_system_score_codex":0.0016629812,"about_ca_system_score_gemma":0.004127552,"threshold_uncertainty_score":0.038015902},"labels":[],"label_agreement":null},{"id":"W4403536220","doi":"10.1145/3691620.3695527","title":"Test-Driven Development and LLM-based Code Generation","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Test (biology); Code generation; Test-driven development; Programming language; Code (set theory); Software engineering; Software development; Operating system; Software","score_opus":0.04352896262435825,"score_gpt":0.27534288710410176,"score_spread":0.23181392447974353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403536220","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06004842,0.00018932878,0.9193825,0.000763145,0.0000650865,0.00066949124,0.0002531654,0.0120635545,0.0065652677],"genre_scores_gemma":[0.30354026,0.0001487969,0.6912471,0.0004156629,0.000014754345,0.00077110867,0.0006032151,0.0011446815,0.0021143788],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9911054,0.0047400985,0.00047822186,0.0007437387,0.0026166749,0.00031583945],"domain_scores_gemma":[0.9702809,0.01920196,0.0017873794,0.0054538543,0.0028788045,0.00039712572],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005964712,0.0007259491,0.0003246905,0.0011421578,0.00037109948,0.0012185536,0.002355548,0.0011748972,0.0026857124],"category_scores_gemma":[0.04253117,0.00059036963,0.0006618479,0.0006851684,0.0016878695,0.0015570675,0.0021243242,0.0013615438,0.00096497184],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072028063,0.0011466037,0.012132542,0.0013677216,0.00013831952,0.0010277745,0.002024284,0.17742285,0.100159384,0.05872488,0.009421371,0.63571405],"study_design_scores_gemma":[0.0002609845,0.0011308942,0.0029544525,0.0002804839,0.00006666669,0.0010305524,0.00024763326,0.78818965,0.14602417,0.030096412,0.02960835,0.00010974448],"about_ca_topic_score_codex":0.0016737378,"about_ca_topic_score_gemma":0.0017445717,"teacher_disagreement_score":0.005964712,"about_ca_system_score_codex":0.001151529,"about_ca_system_score_gemma":0.0018720556,"threshold_uncertainty_score":0.031544805},"labels":[],"label_agreement":null},{"id":"W4403536584","doi":"10.1145/3691620.3695269","title":"Towards a Robust Waiting Strategy for Web GUI Testing for an Industrial Software System","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland; University of Waterloo","funders":"","keywords":"Computer science; Software testing; Software engineering; Software; Web testing; Operating system; The Internet; Web application security; Web development","score_opus":0.2161010280282333,"score_gpt":0.32922369103425075,"score_spread":0.11312266300601745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403536584","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038424928,0.00016838919,0.9538056,0.00008803861,0.000016333086,0.00014780439,0.00004751529,0.006434203,0.0008671616],"genre_scores_gemma":[0.5139533,0.00010419675,0.4835755,0.0001634674,0.000017270117,0.00020485149,0.00024190937,0.0006919335,0.0010476488],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99544406,0.0012596671,0.00040372767,0.0008760512,0.001625222,0.0003912878],"domain_scores_gemma":[0.99178785,0.0035210229,0.0012035508,0.001686741,0.0014464202,0.00035441015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003292469,0.001312077,0.0007165467,0.001889929,0.00038700431,0.0013614447,0.0022671614,0.0011538205,0.0017118378],"category_scores_gemma":[0.010658146,0.000518668,0.0006784949,0.00057944655,0.0010326102,0.001874934,0.0012648712,0.0010599373,0.00077059964],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008911089,0.00064834685,0.0087125,0.0004860846,0.00011401049,0.00109036,0.0007985362,0.17388454,0.3113124,0.018761283,0.0028033382,0.4804975],"study_design_scores_gemma":[0.00006430127,0.00068803295,0.0017654155,0.000053606014,0.00006631755,0.00044330323,0.00010776357,0.90548944,0.081203446,0.007278476,0.0027638627,0.000076009965],"about_ca_topic_score_codex":0.0026023076,"about_ca_topic_score_gemma":0.0020278234,"teacher_disagreement_score":0.003292469,"about_ca_system_score_codex":0.00085760234,"about_ca_system_score_gemma":0.0013655118,"threshold_uncertainty_score":0.017412484},"labels":[],"label_agreement":null},{"id":"W4403536618","doi":"10.1145/3691620.3695261","title":"The Importance of Accounting for Execution Failures when Predicting Test Flakiness","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Computer science; Test (biology); Reliability engineering; Accounting; Engineering; Business","score_opus":0.01749802715021794,"score_gpt":0.2739698035962184,"score_spread":0.2564717764460005,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403536618","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5304784,0.005900101,0.44356778,0.0037180923,0.00037379292,0.0004070749,0.0033943101,0.006636306,0.0055241697],"genre_scores_gemma":[0.9318133,0.00069315586,0.06485224,0.00022584625,0.00015734283,0.000071041824,0.0015749441,0.0001758064,0.00043632195],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.992199,0.002971416,0.0007984788,0.0015448448,0.0019706024,0.00051562174],"domain_scores_gemma":[0.7917214,0.16785581,0.017856767,0.011948194,0.008146539,0.002471264],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010709453,0.0024198534,0.001364495,0.005396304,0.0006914041,0.0026885478,0.0014235454,0.0018074886,0.0013433888],"category_scores_gemma":[0.103620805,0.0004988708,0.00076242466,0.0031739362,0.00086476153,0.0057714074,0.0012258401,0.003141377,0.0009704262],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034080978,0.0003453481,0.6531615,0.00043633464,0.00024239463,0.00024080116,0.0003262199,0.1275582,0.0032598516,0.0013886128,0.0035235581,0.2091764],"study_design_scores_gemma":[0.000030503561,0.00063943275,0.14803109,0.00033403703,0.0002020529,0.0009740175,0.00048513073,0.8289295,0.006286752,0.010435005,0.0035259915,0.0001265072],"about_ca_topic_score_codex":0.008366936,"about_ca_topic_score_gemma":0.011741308,"teacher_disagreement_score":0.010709453,"about_ca_system_score_codex":0.00066941296,"about_ca_system_score_gemma":0.002065883,"threshold_uncertainty_score":0.056637645},"labels":[],"label_agreement":null},{"id":"W4403537012","doi":"10.1145/3691620.3695363","title":"Slicer4D: A Slicing-based Debugger for Java","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Debugger; Computer science; Slicing; Java; Programming language; Program slicing; Operating system; Debugging; World Wide Web","score_opus":0.021165357591978955,"score_gpt":0.29685853510375787,"score_spread":0.2756931775117789,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403537012","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009671522,0.00093361957,0.70382905,0.00020465262,0.000175328,0.00014956335,0.0020507374,0.27894875,0.004036759],"genre_scores_gemma":[0.19564113,0.0012259639,0.7299819,0.00072990137,0.000090094305,0.00043090063,0.009588514,0.052494284,0.009817332],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99895346,0.00022078532,0.00012529561,0.0001928133,0.00040066615,0.0001069568],"domain_scores_gemma":[0.9975683,0.001134633,0.00023653015,0.00055406475,0.0003605619,0.00014590222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001981123,0.0011225005,0.00068761717,0.0014195727,0.00034783606,0.0010958774,0.0020395867,0.0009846353,0.008432698],"category_scores_gemma":[0.005470831,0.0010185554,0.0008004027,0.0006392341,0.0008177274,0.00277368,0.0019809513,0.0019937227,0.0028942768],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002795027,0.00027641095,0.010152757,0.001577075,0.0002967577,0.0010288318,0.0016630874,0.022822142,0.092468254,0.026201067,0.14809704,0.6926215],"study_design_scores_gemma":[0.0011694466,0.0007403062,0.006815902,0.0006181834,0.0002819119,0.0024262439,0.00027989788,0.41950452,0.17633004,0.049087748,0.34221444,0.0005314234],"about_ca_topic_score_codex":0.002473736,"about_ca_topic_score_gemma":0.0047764257,"teacher_disagreement_score":0.008432698,"about_ca_system_score_codex":0.0005457855,"about_ca_system_score_gemma":0.0014797407,"threshold_uncertainty_score":0.028210163},"labels":[],"label_agreement":null},{"id":"W4403576996","doi":"10.48550/arxiv.2410.11769","title":"Can Search-Based Testing with Pareto Optimization Effectively Cover Failure-Revealing Test Inputs?","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Cover (algebra); Pareto principle; Test (biology); Computer science; Pareto optimal; Reliability engineering; Mathematical optimization; Multi-objective optimization; Mathematics; Engineering; Geology; Paleontology","score_opus":0.05888171160340358,"score_gpt":0.19741529985547052,"score_spread":0.13853358825206694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403576996","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18293734,0.0007288352,0.80715805,0.0011031178,0.000059892795,0.00014230827,0.00013205771,0.0010084904,0.0067299483],"genre_scores_gemma":[0.85395503,0.00020923621,0.144125,0.00025553143,0.000018666227,0.00015952364,0.00017621691,0.00014179859,0.00095907674],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99780375,0.0010189784,0.00008966099,0.00023682459,0.00056929444,0.00028142743],"domain_scores_gemma":[0.9911582,0.0065001696,0.0005018819,0.00092665054,0.0007215557,0.00019150894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038754002,0.0011215872,0.0010238313,0.001131002,0.0004252925,0.0009176395,0.0014397688,0.0011401897,0.001805676],"category_scores_gemma":[0.021697177,0.00039494765,0.0009776601,0.0008663908,0.0014381169,0.0020050586,0.0010004067,0.001161343,0.00035167538],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014504183,0.00014144609,0.0042078095,0.00013187507,0.00008057564,0.00009278605,0.0000847138,0.9022204,0.0028447118,0.011120653,0.00091995666,0.07801003],"study_design_scores_gemma":[0.000029155477,0.000109772314,0.0005618464,0.000028123522,0.000018380098,0.00003087569,0.00004175485,0.98295,0.0015819853,0.014113608,0.00052865694,0.000005921555],"about_ca_topic_score_codex":0.0047286856,"about_ca_topic_score_gemma":0.0044994415,"teacher_disagreement_score":0.0047286856,"about_ca_system_score_codex":0.0010263138,"about_ca_system_score_gemma":0.0020455837,"threshold_uncertainty_score":0.020495355},"labels":[],"label_agreement":null},{"id":"W4403935634","doi":"10.1145/3652620.3687793","title":"Concretize: A Model-Driven Tool for Scenario-Based Autonomous Vehicle Testing","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Systems engineering; Engineering","score_opus":0.043968900466269646,"score_gpt":0.2870173616240917,"score_spread":0.24304846115782205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403935634","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036339283,0.000106793326,0.90435755,0.00013202253,0.000043646865,0.0002351833,0.0008898621,0.08761687,0.0029841405],"genre_scores_gemma":[0.15956481,0.0007149897,0.79799366,0.00060360983,0.000060886956,0.0015029288,0.0072521004,0.026533937,0.0057731895],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986797,0.0003621128,0.00010816863,0.00016952101,0.0005664437,0.00011398123],"domain_scores_gemma":[0.99707687,0.0020369452,0.000213277,0.00033850354,0.00023261034,0.00010188641],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024542885,0.0019081469,0.0007803913,0.0017208682,0.000453848,0.0017090158,0.0037243862,0.0019248712,0.01045699],"category_scores_gemma":[0.0072981506,0.0012638873,0.0016862337,0.0005160297,0.0014053327,0.002209669,0.0024494368,0.0025688293,0.0024691785],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00091419666,0.0008042583,0.008165041,0.0023272205,0.0004108603,0.0031081273,0.0015549643,0.39955574,0.05901584,0.14272824,0.10053007,0.28088546],"study_design_scores_gemma":[0.00026497265,0.00019861685,0.00093648105,0.00034747785,0.000058948182,0.0010814167,0.00008946243,0.85254127,0.028195865,0.036287088,0.0798656,0.00013282087],"about_ca_topic_score_codex":0.0027828456,"about_ca_topic_score_gemma":0.0036643145,"teacher_disagreement_score":0.01045699,"about_ca_system_score_codex":0.00087636744,"about_ca_system_score_gemma":0.0016271245,"threshold_uncertainty_score":0.034982085},"labels":[],"label_agreement":null},{"id":"W4404341048","doi":"10.48550/arxiv.2410.21798","title":"Efficient Incremental Code Coverage Analysis for Regression Test Suites","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; University of Waterloo","keywords":"Regression testing; Computer science; Code (set theory); Test (biology); Programming language; Code coverage; Regression analysis; Statistics; Mathematics; Machine learning; Software; Geology","score_opus":0.07111415282567106,"score_gpt":0.22798706258257947,"score_spread":0.15687290975690843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404341048","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13294204,0.00074522634,0.8337394,0.0002996929,0.00005696951,0.00034486613,0.0007344981,0.028265353,0.002872031],"genre_scores_gemma":[0.6149153,0.00022954475,0.37770647,0.00019623838,0.0000730381,0.00047079494,0.0027818445,0.002162027,0.0014648124],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99467963,0.001347855,0.0003145573,0.00073427917,0.0024402582,0.0004833446],"domain_scores_gemma":[0.97542375,0.016258787,0.0021909336,0.002460525,0.0033075565,0.0003585015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002320824,0.0015568451,0.0010038437,0.0056138975,0.00048794592,0.0012532335,0.0021562192,0.0007160471,0.0024132736],"category_scores_gemma":[0.02743589,0.00057036,0.0015330162,0.0020712386,0.00087936944,0.0017621355,0.0016669189,0.0012426943,0.00067836884],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070357573,0.0004574381,0.039939947,0.000756308,0.00027674966,0.00076111604,0.0005425719,0.17259781,0.07973344,0.010790594,0.0077809943,0.68565947],"study_design_scores_gemma":[0.00009815548,0.00029321227,0.007719539,0.00007638584,0.000121662815,0.00039239283,0.00008739792,0.9348929,0.041934524,0.010859341,0.0034776416,0.00004678332],"about_ca_topic_score_codex":0.004575823,"about_ca_topic_score_gemma":0.0056465683,"teacher_disagreement_score":0.0056138975,"about_ca_system_score_codex":0.000996884,"about_ca_system_score_gemma":0.001717307,"threshold_uncertainty_score":0.012273848},"labels":[],"label_agreement":null},{"id":"W4404406203","doi":"10.1007/s10664-024-10564-3","title":"Can search-based testing with pareto optimization effectively cover failure-revealing test inputs?","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; HORIZON EUROPE Reforming and enhancing the European Research and Innovation system; Technische Universität München; European Commission","keywords":"Cover (algebra); Pareto principle; Reliability engineering; Computer science; Engineering; Mathematical optimization; Mathematics; Mechanical engineering","score_opus":0.02105501579744261,"score_gpt":0.25535485620899456,"score_spread":0.23429984041155194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404406203","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42423457,0.00052705785,0.56591296,0.00087128795,0.000056495355,0.000153588,0.000114438946,0.00079714163,0.0073324023],"genre_scores_gemma":[0.9578573,0.00006134636,0.0413006,0.0001070322,0.000008509154,0.00006598467,0.00007140204,0.000048936927,0.00047873225],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983284,0.00073300896,0.00007184039,0.00016808041,0.00043126033,0.000267376],"domain_scores_gemma":[0.9927468,0.005231422,0.00046997963,0.0006399328,0.0006762416,0.00023573873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031034062,0.00093683606,0.0008325371,0.0010681466,0.0004025681,0.00086304685,0.001134423,0.0009469823,0.0016325449],"category_scores_gemma":[0.0141244,0.00031523878,0.00083703536,0.0006115877,0.0011913315,0.0013600314,0.00095183484,0.0009317509,0.00023417424],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000105104664,0.00010228896,0.0029343492,0.00004968298,0.00004001261,0.00006973785,0.000040958275,0.9617073,0.002202952,0.0041609425,0.0003804757,0.028206129],"study_design_scores_gemma":[0.000013902201,0.00008284969,0.00042742642,0.000012445486,0.00000946207,0.000016859985,0.000024842655,0.9940865,0.0012588763,0.0038742945,0.00018896026,0.0000034986965],"about_ca_topic_score_codex":0.005190806,"about_ca_topic_score_gemma":0.0037244658,"teacher_disagreement_score":0.005190806,"about_ca_system_score_codex":0.0011302092,"about_ca_system_score_gemma":0.0016321199,"threshold_uncertainty_score":0.016412556},"labels":[],"label_agreement":null},{"id":"W4404514810","doi":"10.1145/3689944.3696162","title":"BinEq - A Benchmark of Compiled Java Programs to Assess Alternative Builds","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Java; Benchmark (surveying); Programming language; Software engineering; Operating system; Geography; Cartography","score_opus":0.10245218925202816,"score_gpt":0.3377391067043839,"score_spread":0.23528691745235575,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404514810","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9223362,0.000937582,0.042723402,0.00026290875,0.0002176904,0.00040392333,0.009218661,0.016695894,0.007203621],"genre_scores_gemma":[0.8566682,0.00035237757,0.09664621,0.00019769004,0.000050883315,0.000470362,0.0396446,0.003484196,0.0024855065],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953849,0.0013581053,0.0006162119,0.0009381085,0.0012848924,0.0004178905],"domain_scores_gemma":[0.9817074,0.009412682,0.0010341741,0.0043212073,0.0027993254,0.00072518346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004026597,0.0013465849,0.0004608995,0.0024420435,0.00059276476,0.00079422764,0.001810803,0.00095143955,0.0020209788],"category_scores_gemma":[0.017205874,0.0004330994,0.0009271539,0.0023481103,0.0011817017,0.0019460799,0.0016946137,0.0014459119,0.0006978519],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0060179755,0.00500251,0.192807,0.005946917,0.0013872638,0.0017953501,0.0024588546,0.28296664,0.11171177,0.028267106,0.063336164,0.29830235],"study_design_scores_gemma":[0.0007686406,0.005758543,0.13654655,0.00036418388,0.00035379472,0.0013503094,0.0011419686,0.6528737,0.12888923,0.022881197,0.048823327,0.0002485862],"about_ca_topic_score_codex":0.0033824006,"about_ca_topic_score_gemma":0.0032788448,"teacher_disagreement_score":0.004026597,"about_ca_system_score_codex":0.0008838405,"about_ca_system_score_gemma":0.0011604207,"threshold_uncertainty_score":0.021294892},"labels":[],"label_agreement":null},{"id":"W4404860113","doi":"10.1007/s10664-024-10589-8","title":"Contrasting test selection, prioritization, and batch testing at scale","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Concordia University","keywords":"Prioritization; Selection (genetic algorithm); Scale (ratio); Test (biology); Computer science; Regression testing; Reliability engineering; Engineering; Machine learning; Biology; Geography; Management science; Cartography; Operating system; Software","score_opus":0.016660816534096935,"score_gpt":0.2537672231562379,"score_spread":0.23710640662214097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404860113","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8943175,0.0019294876,0.08601288,0.0032795845,0.00012295633,0.00026586722,0.00022239644,0.00064694486,0.013202383],"genre_scores_gemma":[0.98673904,0.00007881837,0.012021151,0.0001860204,0.000047976024,0.00006975466,0.00008790527,0.00006861456,0.00070072734],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.96169186,0.02681852,0.0010594625,0.0031087443,0.0055371625,0.0017843539],"domain_scores_gemma":[0.45321968,0.49802822,0.015919702,0.0189081,0.010157303,0.0037669253],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.046024665,0.0012004359,0.0014860422,0.0018941776,0.0010430013,0.0030197862,0.003107627,0.0020464475,0.0038235756],"category_scores_gemma":[0.28283623,0.00060154434,0.000863053,0.0022393207,0.0040273466,0.0070025884,0.0023197555,0.0026111295,0.00036534842],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013953913,0.0041357824,0.28279185,0.0011102263,0.0016324478,0.00083405577,0.0034516463,0.1524799,0.013311325,0.111899495,0.007232205,0.40716714],"study_design_scores_gemma":[0.0023917828,0.0069831647,0.32130638,0.00028773525,0.0011905495,0.0005499144,0.0031692986,0.3992859,0.00986393,0.25047982,0.004226229,0.000265319],"about_ca_topic_score_codex":0.011993962,"about_ca_topic_score_gemma":0.013060074,"teacher_disagreement_score":0.046024665,"about_ca_system_score_codex":0.0029649367,"about_ca_system_score_gemma":0.0035841744,"threshold_uncertainty_score":0.24340463},"labels":[],"label_agreement":null},{"id":"W4405094850","doi":"10.48550/arxiv.2412.03843","title":"Using Cooperative Co-evolutionary Search to Generate Metamorphic Test Cases for Autonomous Driving Systems","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Metamorphic rock; Test (biology); Computer science; Artificial intelligence; Geology; Paleontology","score_opus":0.2180923605641594,"score_gpt":0.27285307060026287,"score_spread":0.05476071003610347,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405094850","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5044192,0.00063817145,0.4836603,0.00065193285,0.0000934637,0.00035986264,0.00024824895,0.0029548116,0.0069739176],"genre_scores_gemma":[0.8796749,0.00007823899,0.11827503,0.00015937786,0.00001200065,0.00017504551,0.00035625708,0.00014741602,0.0011217629],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989517,0.00041761267,0.00004882375,0.00016699545,0.00028837915,0.0001263969],"domain_scores_gemma":[0.9949949,0.0036769176,0.00026983686,0.00035395287,0.00052904757,0.00017534275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001534584,0.0011891082,0.0005119028,0.0014429903,0.00037775224,0.00066356704,0.0017973277,0.0011277996,0.0015269615],"category_scores_gemma":[0.0077834046,0.0004198828,0.0006922236,0.00050845183,0.0010052726,0.00076632935,0.0012844524,0.00090028625,0.00019624214],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009851458,0.00018182353,0.006236514,0.00008341988,0.00005321792,0.00028618993,0.000105048304,0.92511755,0.004838897,0.003061344,0.0008787625,0.059058763],"study_design_scores_gemma":[0.000014776145,0.000049458144,0.00021291118,0.0000054833563,0.0000071331037,0.000021713686,0.000016801598,0.9970073,0.0010491086,0.0013240295,0.00028797236,0.0000033183655],"about_ca_topic_score_codex":0.006375893,"about_ca_topic_score_gemma":0.0082730455,"teacher_disagreement_score":0.006375893,"about_ca_system_score_codex":0.00087293005,"about_ca_system_score_gemma":0.0013072671,"threshold_uncertainty_score":0.01267755},"labels":[],"label_agreement":null},{"id":"W4405850275","doi":"10.36548/jitdw.2024.4.002","title":"Increasing Clustering Efficiency with QRDSO and WAC-HACK: A Hybrid Optimization Framework in Software Testing","year":2024,"lang":"en","type":"article","venue":"Journal of Information Technology and Digital World","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"PotashCorp (Canada)","funders":"","keywords":"Cluster analysis; Software testing; Computer science; Software; Artificial intelligence; Programming language","score_opus":0.0071124343636870474,"score_gpt":0.22081059287896276,"score_spread":0.2136981585152757,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405850275","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019733142,0.0001564812,0.9784718,0.00018019139,0.000026259568,0.00004458446,0.000023149822,0.00048411146,0.00088029914],"genre_scores_gemma":[0.5638651,0.00012884935,0.4338776,0.00018925633,0.00003218289,0.0001741248,0.00014522637,0.00017908402,0.0014085632],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987664,0.00047572554,0.00007780706,0.00018251482,0.0003930956,0.00010446108],"domain_scores_gemma":[0.99827313,0.00086021103,0.00015153943,0.00020853568,0.00041184868,0.0000946934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001967239,0.0006203543,0.0010942197,0.0012805326,0.0006143276,0.00084861915,0.0015175742,0.0009571575,0.0010550141],"category_scores_gemma":[0.004683818,0.00038967407,0.0006304745,0.0014242817,0.0012712077,0.001497138,0.0017239872,0.0008910864,0.00021438184],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011579283,0.00012110729,0.002656419,0.000120693585,0.0000716434,0.00005490801,0.0001119331,0.7557165,0.006672097,0.03517398,0.0016994191,0.19748561],"study_design_scores_gemma":[0.0000069829453,0.000023533214,0.00015403955,0.0000025739384,0.000003400283,0.000010221129,0.000008800515,0.9944689,0.00081862434,0.0042285323,0.00026852117,0.0000057371167],"about_ca_topic_score_codex":0.003962944,"about_ca_topic_score_gemma":0.004253276,"teacher_disagreement_score":0.003962944,"about_ca_system_score_codex":0.0011698768,"about_ca_system_score_gemma":0.0015125868,"threshold_uncertainty_score":0.010403872},"labels":[],"label_agreement":null},{"id":"W4406460194","doi":"10.1109/tps-isa62245.2024.00043","title":"Fine-Tuning LLMs for Code Mutation: A New Era of Cyber Threats","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Computer security; Code (set theory); Mutation; Internet privacy; Programming language; Genetics; Biology","score_opus":0.04922846117780114,"score_gpt":0.323486207207921,"score_spread":0.2742577460301198,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406460194","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20587733,0.0014025986,0.7609706,0.0016798822,0.0002534595,0.00017797457,0.0002923679,0.026156366,0.0031894846],"genre_scores_gemma":[0.7565336,0.0004153366,0.2366382,0.0009351372,0.00007494829,0.00019533365,0.0007608033,0.0016909515,0.0027556717],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986519,0.00039671513,0.00008287305,0.0003713459,0.00039032236,0.000106849075],"domain_scores_gemma":[0.99623764,0.0018469174,0.0003696989,0.0009805749,0.0004360093,0.00012923291],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016978937,0.0011514648,0.0006949453,0.0006231804,0.0003873893,0.0010113086,0.0016902367,0.0012369892,0.0012488128],"category_scores_gemma":[0.010553426,0.0005223413,0.000787362,0.0002794064,0.0011834905,0.0021821412,0.001409281,0.0025344447,0.000892844],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036025562,0.0003515501,0.008338969,0.00035290865,0.00012332015,0.0003633108,0.00050240656,0.5449536,0.07068791,0.0065489155,0.004861022,0.36255586],"study_design_scores_gemma":[0.000019381161,0.00014249042,0.00049857754,0.00003063869,0.000021115236,0.00008973372,0.000041532418,0.97057885,0.019734489,0.0051953318,0.003626006,0.000021929252],"about_ca_topic_score_codex":0.0032436904,"about_ca_topic_score_gemma":0.003728683,"teacher_disagreement_score":0.0032436904,"about_ca_system_score_codex":0.00089345756,"about_ca_system_score_gemma":0.0016307529,"threshold_uncertainty_score":0.00897944},"labels":[],"label_agreement":null},{"id":"W4406688914","doi":"10.1016/b978-0-08-057206-2.50019-x","title":"10.1016/b978-0-08-057206-2.50019-x","year":2000,"lang":"en","type":"book-chapter","venue":"Time to knit","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.0132082003442451,"score_gpt":0.19251967506130138,"score_spread":0.17931147471705627,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406688914","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00021451959,0.00042226972,0.0012402418,0.0003040755,0.00014689848,0.00004356638,0.00061569805,0.0010113696,0.9960014],"genre_scores_gemma":[0.0005665249,0.00026530595,0.00047643806,0.000120726225,0.000032990236,0.000034565033,0.00041381142,0.00027402333,0.9978156],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995433,0.000028801778,0.000028921813,0.00016516654,0.00015347074,0.00008031187],"domain_scores_gemma":[0.9982393,0.00055082364,0.00013560872,0.00026348874,0.00033210186,0.00047864346],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0009255899,0.0021866674,0.0013708083,0.0018161652,0.0013372246,0.004771358,0.0027534072,0.0045754733,0.9772752],"category_scores_gemma":[0.0017218213,0.0008412143,0.00078404223,0.0021677816,0.0012252821,0.005158847,0.0028459285,0.0022340987,0.9884774],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000121532794,0.000118230986,0.00043648266,0.00046167997,0.000019203551,0.00013511587,0.00010691603,0.00038057516,0.0025704128,0.009146941,0.35259002,0.6339129],"study_design_scores_gemma":[0.000015849808,0.000045867975,0.00050745974,0.00026433307,0.000008571851,0.00019251286,0.000079780766,0.0001364368,0.0003489671,0.0013851371,0.9970022,0.000012833974],"about_ca_topic_score_codex":0.002757161,"about_ca_topic_score_gemma":0.0026574288,"teacher_disagreement_score":0.022724807,"about_ca_system_score_codex":0.0008673221,"about_ca_system_score_gemma":0.0007893167,"threshold_uncertainty_score":0.032414198},"labels":[],"label_agreement":null},{"id":"W4406866174","doi":"10.1145/3715008","title":"Scalable Similarity-Aware Test Suite Minimization with Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Test suite; Reinforcement learning; Suite; Scalability; Minification; Similarity (geometry); Test (biology); Artificial intelligence; Software engineering; Machine learning; Test case; Programming language; Operating system","score_opus":0.04087899492526334,"score_gpt":0.2917156633049735,"score_spread":0.2508366683797102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406866174","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07953009,0.00064597087,0.909829,0.000604361,0.000058195175,0.00030131187,0.0002484203,0.0058884732,0.0028941503],"genre_scores_gemma":[0.7348679,0.00014360348,0.26067922,0.00041369797,0.000056868877,0.00044504934,0.0010869538,0.0005576891,0.0017490966],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99798125,0.0007610918,0.00009571987,0.00040095206,0.00053355127,0.00022738292],"domain_scores_gemma":[0.9937338,0.0044193217,0.00051197724,0.0005177365,0.0005641155,0.00025301834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019303793,0.0017302461,0.001463363,0.0011304257,0.0003786787,0.00076562155,0.002193855,0.0011019119,0.0026208963],"category_scores_gemma":[0.009485103,0.0006529322,0.0011574406,0.0008268983,0.0010692684,0.0013616671,0.001498418,0.0021044281,0.00053345965],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001349174,0.00028871695,0.0021443788,0.000179478,0.00006627223,0.00012526775,0.000055400043,0.85406226,0.0044277934,0.0027284687,0.0029034482,0.13288364],"study_design_scores_gemma":[0.000024971869,0.00005504497,0.00013529489,0.000006006191,0.00000844138,0.000020342357,0.000008706248,0.99590826,0.00077976397,0.0028293168,0.00022052914,0.0000032358234],"about_ca_topic_score_codex":0.0038278026,"about_ca_topic_score_gemma":0.0049206633,"teacher_disagreement_score":0.0038278026,"about_ca_system_score_codex":0.001221734,"about_ca_system_score_gemma":0.001960493,"threshold_uncertainty_score":0.010208964},"labels":[],"label_agreement":null},{"id":"W4407375771","doi":"10.1109/tse.2025.3541166","title":"Automated Test Case Repair Using Language Models","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of Ottawa","funders":"","keywords":"Computer science; Programming language; Test (biology); Software engineering; Model-based testing; Reliability engineering; Test case; Natural language processing; Machine learning","score_opus":0.016937912449282302,"score_gpt":0.264907490693648,"score_spread":0.2479695782443657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407375771","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035338785,0.00016134627,0.9407604,0.00024963004,0.000050234055,0.00014794392,0.000249248,0.02088972,0.0021527058],"genre_scores_gemma":[0.6549723,0.00012349764,0.340464,0.0001225707,0.000020667205,0.00020411245,0.0006969121,0.0015590186,0.001836851],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974821,0.0009828376,0.000154993,0.00034893522,0.00079894,0.00023221651],"domain_scores_gemma":[0.9899734,0.0062434985,0.0007479583,0.0018214639,0.0010640683,0.00014961469],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014084746,0.0013323956,0.0009020319,0.0015323223,0.00050459104,0.0017537269,0.0021241787,0.0011914746,0.004174627],"category_scores_gemma":[0.010534418,0.00081021787,0.0015802844,0.0006877905,0.00075698126,0.0026998613,0.0015365522,0.0012293145,0.0011456972],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008766106,0.0006637577,0.005510743,0.0005797607,0.00023226097,0.0012904131,0.000620761,0.488429,0.038106255,0.037867084,0.01009853,0.41572475],"study_design_scores_gemma":[0.00004460781,0.000058524514,0.00017095827,0.00002826341,0.000042109707,0.000099851924,0.000044860833,0.97653174,0.008725635,0.012571926,0.0016633746,0.000018109242],"about_ca_topic_score_codex":0.005527335,"about_ca_topic_score_gemma":0.008360042,"teacher_disagreement_score":0.005527335,"about_ca_system_score_codex":0.0010517967,"about_ca_system_score_gemma":0.0021277415,"threshold_uncertainty_score":0.0139654875},"labels":[],"label_agreement":null},{"id":"W4407416591","doi":"10.1007/978-3-031-81573-7_6","title":"Leveraging Conversational AI for Accelerating User-Driven Software Testing","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes of the Institute for Computer Sciences, Social Informatics and Telecommunications Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Software testing; Human–computer interaction; Software; Software engineering; Programming language","score_opus":0.048905755139386414,"score_gpt":0.26813146808301624,"score_spread":0.21922571294362983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407416591","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016436893,0.00029063824,0.95764947,0.0003001785,0.00020735925,0.00012765874,0.000080405174,0.0146478005,0.010259615],"genre_scores_gemma":[0.38525674,0.00021908815,0.60256666,0.0003886629,0.00013615316,0.0002069447,0.00027885454,0.0016595648,0.009287364],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99704933,0.0012079924,0.00013759921,0.00044145036,0.00091389683,0.0002496428],"domain_scores_gemma":[0.9912572,0.0063029374,0.00022862738,0.001207483,0.0007505875,0.00025319058],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019592675,0.0009695049,0.0006732589,0.00089243654,0.0007122023,0.0021715548,0.0025063488,0.0012227875,0.0084177805],"category_scores_gemma":[0.011261114,0.000618449,0.0006856491,0.00049041543,0.0011124929,0.0035805271,0.0028314136,0.0028144442,0.002697595],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075986655,0.0007987246,0.0026382005,0.0004617636,0.00015867224,0.0005626878,0.0017480341,0.04164641,0.10509399,0.060262788,0.0139877265,0.7718811],"study_design_scores_gemma":[0.000049475486,0.00016700813,0.00030950358,0.000043828422,0.00006242655,0.00023335211,0.00017759671,0.89146394,0.034239803,0.056665625,0.016539313,0.000048095084],"about_ca_topic_score_codex":0.002040979,"about_ca_topic_score_gemma":0.003108156,"teacher_disagreement_score":0.0084177805,"about_ca_system_score_codex":0.0004937598,"about_ca_system_score_gemma":0.0010544263,"threshold_uncertainty_score":0.028160274},"labels":[],"label_agreement":null},{"id":"W4407639620","doi":"10.1109/acit62805.2024.10877022","title":"A Data-Driven Approach Towards Software Regression Testing Quality Optimization","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Becton Dickinson (Canada)","funders":"","keywords":"Regression testing; Computer science; Quality (philosophy); Software quality; Software testing; Regression analysis; Software; Reliability engineering; Machine learning; Software development; Software construction; Programming language; Engineering","score_opus":0.17284480908161898,"score_gpt":0.3693146945969045,"score_spread":0.19646988551528555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407639620","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004852357,0.0001684062,0.9933031,0.0002312714,0.000014599137,0.000085862834,0.00010893474,0.0006228505,0.0006125316],"genre_scores_gemma":[0.2455703,0.00035529907,0.7511486,0.00021217902,0.00008341283,0.0004634535,0.0006180502,0.00033978958,0.0012089751],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942398,0.0021337755,0.0003110582,0.00084479415,0.0021669352,0.00030369172],"domain_scores_gemma":[0.98617095,0.008781522,0.0011678184,0.0011864526,0.0024937703,0.00019941098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057076467,0.00155784,0.0016268202,0.0026959262,0.0004862021,0.0021382875,0.0029859317,0.0013172256,0.0016344276],"category_scores_gemma":[0.0208227,0.0009946248,0.0015777015,0.0022743694,0.001196648,0.0018122386,0.0015371031,0.0024370167,0.00048152902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000103141865,0.00021270779,0.002277657,0.00020000103,0.000119838274,0.00012037537,0.0000789325,0.8632485,0.00445434,0.015377557,0.0010825866,0.11272435],"study_design_scores_gemma":[0.000009358543,0.00003876808,0.00019359913,0.000016557857,0.000013468542,0.00002060562,0.000008739309,0.9911067,0.001770621,0.006220837,0.0005928471,0.000007933704],"about_ca_topic_score_codex":0.003994998,"about_ca_topic_score_gemma":0.003980659,"teacher_disagreement_score":0.0057076467,"about_ca_system_score_codex":0.0015271577,"about_ca_system_score_gemma":0.0021415295,"threshold_uncertainty_score":0.030185282},"labels":[],"label_agreement":null},{"id":"W4407771524","doi":"10.1145/3641554.3701809","title":"How Effective and Efficient are Student-Written Software Tests?","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Universitas Brawijaya","keywords":"Computer science; Software testing; Software; Software engineering; Programming language","score_opus":0.008202861097243937,"score_gpt":0.26587968550532937,"score_spread":0.2576768244080854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407771524","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9560221,0.0031001575,0.027311858,0.003772341,0.00012588213,0.00018716644,0.00027885693,0.00060949044,0.008592112],"genre_scores_gemma":[0.9831777,0.00068954105,0.014380812,0.0003680285,0.000063876476,0.000058781938,0.00024074932,0.00010755707,0.0009128643],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9472227,0.033135977,0.0038734165,0.002762429,0.011522889,0.0014825066],"domain_scores_gemma":[0.64535505,0.25337198,0.051860377,0.014993393,0.03006324,0.0043559587],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023855284,0.0006533524,0.00075408514,0.0027357072,0.00029307173,0.0039829775,0.0012080007,0.0012156392,0.001467001],"category_scores_gemma":[0.27463815,0.00029710095,0.00041345326,0.0021731756,0.0014109839,0.005798251,0.0011108674,0.00082923664,0.00096549466],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045536633,0.00046954825,0.37237146,0.00039783074,0.0001504041,0.00020728575,0.0019612755,0.0022983435,0.0031743678,0.0008012973,0.0022384354,0.61547446],"study_design_scores_gemma":[0.00021797627,0.0035779227,0.8957169,0.0018034481,0.00036831008,0.0028355638,0.009583189,0.036000025,0.02444247,0.00795675,0.017322095,0.00017542484],"about_ca_topic_score_codex":0.001208891,"about_ca_topic_score_gemma":0.0020042076,"teacher_disagreement_score":0.023855284,"about_ca_system_score_codex":0.00068557105,"about_ca_system_score_gemma":0.0015642704,"threshold_uncertainty_score":0.12616032},"labels":[],"label_agreement":null},{"id":"W4407848466","doi":"10.1145/3696443.3708929","title":"LLM-Vectorizer: LLM-Based Verified Loop Vectorizer","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Loop (graph theory); Mathematics","score_opus":0.011359170845947314,"score_gpt":0.26283494927821155,"score_spread":0.2514757784322642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407848466","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.080154724,0.0014901299,0.5614663,0.00063923106,0.000505496,0.0004354835,0.0060661915,0.34146953,0.0077729193],"genre_scores_gemma":[0.4134844,0.0004119018,0.54006976,0.0007961137,0.00010207091,0.00057375064,0.023690505,0.015430256,0.0054412563],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971552,0.000691086,0.000259101,0.0006774633,0.00094783533,0.00026922516],"domain_scores_gemma":[0.99432075,0.0018606504,0.00050573255,0.0021376887,0.0010676585,0.000107516404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025324144,0.002421226,0.00067585736,0.0014801389,0.00071594166,0.0014880197,0.0036915725,0.0011609498,0.0077565494],"category_scores_gemma":[0.010425331,0.0010039127,0.0014821424,0.0008891158,0.0014923272,0.0036353495,0.0020515136,0.0019126249,0.0032775968],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024495737,0.00068242993,0.02907348,0.0023031395,0.0005063502,0.00063706236,0.00058869435,0.22374801,0.058699448,0.027169598,0.15746836,0.4966738],"study_design_scores_gemma":[0.00039443592,0.0005365126,0.0017411513,0.00013214191,0.00009847449,0.00021362418,0.00010944568,0.8694706,0.07971021,0.016210396,0.03130767,0.000075398246],"about_ca_topic_score_codex":0.0055544004,"about_ca_topic_score_gemma":0.011032253,"teacher_disagreement_score":0.0077565494,"about_ca_system_score_codex":0.0012797797,"about_ca_system_score_gemma":0.0039650206,"threshold_uncertainty_score":0.025948286},"labels":[],"label_agreement":null},{"id":"W4408120458","doi":"10.5220/0013186800003896","title":"On the Generation of Input Space Model for Model-Driven Requirements-Based Testing","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Computer science; Space (punctuation); Model-based testing; Systems engineering; Engineering; Operating system; Test case","score_opus":0.166819225024256,"score_gpt":0.33517492766224016,"score_spread":0.16835570263798416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408120458","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049748938,0.000042286636,0.9925774,0.00006265535,0.000012819454,0.00006316573,0.00007873602,0.0011010958,0.0010871049],"genre_scores_gemma":[0.31036007,0.00017702574,0.68535596,0.00013629746,0.000019829906,0.00024142407,0.0009240064,0.0009210091,0.0018643651],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99847347,0.00058757956,0.000078144185,0.000149376,0.0005951133,0.000116219046],"domain_scores_gemma":[0.9962005,0.0023552305,0.00016714717,0.0006643473,0.00054733676,0.00006549473],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010991603,0.0007988727,0.00076634716,0.0011059712,0.00045092255,0.0012471178,0.0014605348,0.00086192746,0.004421406],"category_scores_gemma":[0.0078047775,0.0006083203,0.0015770692,0.00072972546,0.0007086097,0.0014891259,0.0014485795,0.0012173267,0.0008227077],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039063825,0.00025506882,0.0021119283,0.00042498662,0.0001003563,0.0006218089,0.00043349518,0.52405757,0.024700684,0.119430415,0.0056263604,0.32184666],"study_design_scores_gemma":[0.00002119577,0.00003275497,0.00012790221,0.000028548071,0.000019583731,0.000077210454,0.00001918772,0.97311157,0.005185905,0.019661726,0.0017038641,0.000010591141],"about_ca_topic_score_codex":0.005365029,"about_ca_topic_score_gemma":0.005953044,"teacher_disagreement_score":0.005365029,"about_ca_system_score_codex":0.0007510912,"about_ca_system_score_gemma":0.0011131772,"threshold_uncertainty_score":0.014791071},"labels":[],"label_agreement":null},{"id":"W4408729936","doi":"10.63485/waect-f7t64","title":"Beta testers needed for Canadian public domain registry","year":2008,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Public domain; Domain (mathematical analysis); Business; Computer science; Geography; Mathematics","score_opus":0.056251919102948654,"score_gpt":0.2695010204006764,"score_spread":0.21324910129772776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408729936","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033344597,0.001978925,0.105736405,0.032481056,0.004845334,0.0035697098,0.03214656,0.12582469,0.6600728],"genre_scores_gemma":[0.18929414,0.0028624071,0.15718362,0.0071414346,0.0012988909,0.00246141,0.07563706,0.03226353,0.5318576],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98178506,0.0024079552,0.0009928909,0.0016184611,0.009256407,0.0039392533],"domain_scores_gemma":[0.867999,0.008092406,0.0025956025,0.018044267,0.08868094,0.014587678],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.020459589,0.0012463601,0.0011344888,0.006394138,0.0059140115,0.006092931,0.004318313,0.0018641416,0.13925007],"category_scores_gemma":[0.056140587,0.0017571562,0.00076785235,0.0058847805,0.0014229056,0.007907456,0.004379498,0.0037082743,0.057900682],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003570976,0.00022386739,0.005327601,0.00013301568,0.000011096553,0.00063097867,0.0014290363,0.00044751944,0.0042561567,0.024263602,0.8141725,0.14874744],"study_design_scores_gemma":[0.00009647737,0.000055404194,0.0037953774,0.00013835913,0.00001574148,0.00026353507,0.0007544156,0.00093750027,0.0026953518,0.0020053845,0.9891629,0.000079402664],"about_ca_topic_score_codex":0.59566325,"about_ca_topic_score_gemma":0.6299977,"teacher_disagreement_score":0.9956817,"about_ca_system_score_codex":0.016967207,"about_ca_system_score_gemma":0.07225416,"threshold_uncertainty_score":0.8134359},"labels":[],"label_agreement":null},{"id":"W4408734932","doi":"10.1007/s42979-025-03800-0","title":"A Mechanized Method for Risk-Based Test Case Generation and Prioritization by an Improved Correlation System","year":2025,"lang":"en","type":"article","venue":"SN Computer Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Tornado Spectral Systems (Canada)","funders":"","keywords":"Prioritization; Test (biology); Reliability engineering; Computer science; Correlation; Risk analysis (engineering); Statistics; Engineering; Mathematics; Geology; Business; Management science","score_opus":0.012951214921615752,"score_gpt":0.28994047690220215,"score_spread":0.2769892619805864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408734932","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023956166,0.00002815566,0.98774964,0.00003290707,0.000019908884,0.00012830315,0.00008891249,0.0091162175,0.00044040722],"genre_scores_gemma":[0.058436047,0.000032150303,0.93893063,0.00009377008,0.000027340046,0.00026675928,0.000320839,0.0006639667,0.0012285546],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961702,0.0010556877,0.00036443112,0.0007983281,0.0013792117,0.00023206131],"domain_scores_gemma":[0.99179244,0.0044164103,0.0006345007,0.001438709,0.0015421985,0.00017568952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017531335,0.001276916,0.0011938307,0.002929125,0.0005241975,0.0015758459,0.0018912193,0.0008542626,0.015501196],"category_scores_gemma":[0.009501932,0.00096495706,0.0010962548,0.0017716367,0.00062056776,0.0012069616,0.0017437164,0.0016581662,0.0032238737],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045352412,0.00040917823,0.0022867096,0.0003900063,0.00017958682,0.00041255305,0.00021695299,0.04558005,0.047086764,0.012419536,0.011165389,0.8793997],"study_design_scores_gemma":[0.00028252474,0.00029236387,0.002178088,0.0000917227,0.000166124,0.0008464252,0.000051684983,0.938386,0.036113057,0.008963714,0.01251828,0.000109997774],"about_ca_topic_score_codex":0.0048132734,"about_ca_topic_score_gemma":0.0052528055,"teacher_disagreement_score":0.015501196,"about_ca_system_score_codex":0.0008231857,"about_ca_system_score_gemma":0.003113605,"threshold_uncertainty_score":0.051856697},"labels":[],"label_agreement":null},{"id":"W4408851891","doi":"10.1145/3725212","title":"Towards On-the-Fly Code Performance Profiling","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"National Key Research and Development Program of China","keywords":"Computer science; Profiling (computer programming); On the fly; Programming language; Software engineering; Operating system","score_opus":0.088324649823905,"score_gpt":0.3216710607829046,"score_spread":0.23334641095899958,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408851891","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10716114,0.00086750166,0.823279,0.0006251524,0.00008616327,0.0001082478,0.0015733875,0.062389657,0.003909704],"genre_scores_gemma":[0.6123236,0.00050785893,0.37432718,0.00037896066,0.000062738734,0.0001389792,0.005885637,0.0032513726,0.003123648],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987457,0.00020923755,0.000050397593,0.00042968863,0.0004327357,0.00013217745],"domain_scores_gemma":[0.9966865,0.00096678507,0.00048749434,0.0008716003,0.00083991664,0.00014773906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007441768,0.0020095492,0.00068140845,0.0023475238,0.0003306542,0.0011514265,0.0015641594,0.0009985013,0.0012639222],"category_scores_gemma":[0.0074222647,0.00071880035,0.0006798364,0.0010854339,0.00040319742,0.0027112672,0.0011485147,0.0016799659,0.002374214],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037796778,0.0005190811,0.034469154,0.00026452844,0.00009440934,0.0002743736,0.00023676902,0.16975941,0.057150666,0.0038308455,0.017460683,0.71556216],"study_design_scores_gemma":[0.000009049394,0.00005604814,0.0027951524,0.000019863315,0.000012486175,0.00006374733,0.000029106937,0.97517395,0.014270946,0.0043638595,0.0031871663,0.000018595052],"about_ca_topic_score_codex":0.0044529755,"about_ca_topic_score_gemma":0.0072434847,"teacher_disagreement_score":0.0044529755,"about_ca_system_score_codex":0.00066220306,"about_ca_system_score_gemma":0.0013800313,"threshold_uncertainty_score":0.008854091},"labels":[],"label_agreement":null},{"id":"W4409077140","doi":"10.1109/tbdata.2025.3556615","title":"Utility-Driven Data Analytics Algorithm for Transaction Modifications Using Pre-Large Concept With Single Database Scan","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Big Data","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Database transaction; Analytics; Database; Transaction processing; Data mining; Transaction data; Algorithm; Transaction log; Distributed transaction","score_opus":0.2046491776236223,"score_gpt":0.3506634177596448,"score_spread":0.14601424013602252,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409077140","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04734599,0.00033877013,0.9472261,0.0002213486,0.000038810358,0.0003241708,0.0002791006,0.0032785304,0.0009471577],"genre_scores_gemma":[0.3791113,0.00019772779,0.61731994,0.00015405078,0.00003030028,0.00039893689,0.00087789516,0.00016247932,0.0017474272],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99852705,0.00015339778,0.0001309513,0.0003731103,0.00071433774,0.000101103156],"domain_scores_gemma":[0.9971077,0.0009975873,0.00032884613,0.00061805535,0.0008220632,0.00012573284],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012811383,0.00090202544,0.0009769729,0.0028216722,0.000715422,0.0012220534,0.0023933346,0.0005688545,0.0013407149],"category_scores_gemma":[0.0055409134,0.0004801899,0.00090466347,0.0029576425,0.0007664399,0.002725137,0.0015618935,0.0013973714,0.000629562],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046458986,0.00037748492,0.018226322,0.00023798071,0.0001222485,0.0005216516,0.00052636117,0.07138486,0.028379908,0.0143111,0.006551265,0.85889626],"study_design_scores_gemma":[0.00004886886,0.0001866706,0.0018600794,0.000018277999,0.00003184807,0.0006298905,0.00017843943,0.9632093,0.016483132,0.0135178035,0.003804764,0.000030996358],"about_ca_topic_score_codex":0.0025108783,"about_ca_topic_score_gemma":0.0036594067,"teacher_disagreement_score":0.0028216722,"about_ca_system_score_codex":0.00066956517,"about_ca_system_score_gemma":0.001929707,"threshold_uncertainty_score":0.006775379},"labels":[],"label_agreement":null},{"id":"W4409602065","doi":"10.1007/978-3-031-85859-8_13","title":"Cross Language Soccer Framework: An Open Source Framework for the RoboCup 2D Soccer Simulation","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland; Pleiades Robotics (Canada)","funders":"","keywords":"Computer science; Open source; Artificial intelligence; Human–computer interaction; Programming language","score_opus":0.037695864528258786,"score_gpt":0.36416690756772285,"score_spread":0.32647104303946406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409602065","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052530663,0.0004959065,0.7770982,0.00015128721,0.0002421855,0.00033641414,0.0038474388,0.18777616,0.024799393],"genre_scores_gemma":[0.14112237,0.0016165798,0.54770505,0.00093812594,0.00015866378,0.0018392882,0.030230273,0.19940424,0.0769854],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995085,0.00005899456,0.000033169224,0.00006909105,0.0002631965,0.00006702694],"domain_scores_gemma":[0.99954283,0.0001298884,0.000028477454,0.000075543896,0.00015721413,0.00006612221],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007133305,0.0018324605,0.00091029325,0.0010995384,0.00056672527,0.001519726,0.0036416415,0.0013578488,0.053966254],"category_scores_gemma":[0.0016240459,0.0015821839,0.0015902058,0.00062345364,0.00047128645,0.0014006877,0.0026754797,0.0021967879,0.017954433],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007939022,0.00062170124,0.0032667213,0.0018838479,0.0004899294,0.0008751646,0.0010053308,0.18042965,0.042427994,0.048592377,0.34555632,0.37405708],"study_design_scores_gemma":[0.00028224714,0.00020134279,0.0017448213,0.00029218974,0.00012770352,0.0007438364,0.000126786,0.44908437,0.029156903,0.02105844,0.49694368,0.00023779452],"about_ca_topic_score_codex":0.0050967513,"about_ca_topic_score_gemma":0.0074564177,"teacher_disagreement_score":0.053966254,"about_ca_system_score_codex":0.00065797433,"about_ca_system_score_gemma":0.0012638338,"threshold_uncertainty_score":0.18053508},"labels":[],"label_agreement":null},{"id":"W4409728046","doi":"10.1016/j.jss.2025.112453","title":"An empirical evaluation of static, dynamic, and hybrid slicing of WebAssembly binaries","year":2025,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Slicing; Computer science; Computer graphics (images)","score_opus":0.027482611779639187,"score_gpt":0.346174844022441,"score_spread":0.3186922322428018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409728046","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97753215,0.0016172476,0.014532391,0.00015416033,0.000049402184,0.00022505663,0.0009409444,0.0017830853,0.0031655808],"genre_scores_gemma":[0.96314347,0.00040335825,0.031888194,0.000070818955,0.000030881587,0.00014240417,0.0030089077,0.00040190874,0.0009100375],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9908585,0.0035331557,0.00087391795,0.0012107164,0.003111525,0.00041218215],"domain_scores_gemma":[0.8725008,0.09172563,0.008334985,0.014146445,0.011655179,0.0016369735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010666909,0.001160644,0.00058725325,0.0026197908,0.0006737271,0.0011314389,0.0016674458,0.0008726419,0.0016213068],"category_scores_gemma":[0.069851816,0.000564713,0.0006164614,0.001953125,0.0015983902,0.003232609,0.0014122261,0.0011172679,0.0004985725],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0106214,0.0036125442,0.15759839,0.0035484617,0.00086300925,0.00063801237,0.0026711156,0.24940394,0.041619614,0.006208386,0.011870059,0.51134515],"study_design_scores_gemma":[0.00091456523,0.012352344,0.16628471,0.0006342897,0.0006048223,0.0016003218,0.002186294,0.72682273,0.065051414,0.005978745,0.01733368,0.0002360882],"about_ca_topic_score_codex":0.0033686755,"about_ca_topic_score_gemma":0.006629405,"teacher_disagreement_score":0.010666909,"about_ca_system_score_codex":0.0014468831,"about_ca_system_score_gemma":0.00082497176,"threshold_uncertainty_score":0.056412697},"labels":[],"label_agreement":null},{"id":"W4409795042","doi":"10.61091/jcmcc127b-386","title":"RESTful API-based software interface testing techniques and common problems analysis","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Software testing; Software engineering; Interface (matter); Software; Operating system; Programming language","score_opus":0.018204846784951725,"score_gpt":0.2816590800225939,"score_spread":0.26345423323764217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409795042","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021276698,0.00026048126,0.9748079,0.00017972119,0.000015807904,0.00010450045,0.000052485873,0.001544354,0.0017580064],"genre_scores_gemma":[0.4848207,0.0003384834,0.51203126,0.00014763417,0.000031380656,0.00024389081,0.0002377185,0.00023708538,0.0019117403],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9955212,0.0010194348,0.00030228327,0.0007907365,0.0021048777,0.00026156844],"domain_scores_gemma":[0.994142,0.0027948644,0.00063395547,0.0009811659,0.001323892,0.00012417426],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020539754,0.0010732817,0.00072018406,0.004244712,0.00085922104,0.001541915,0.002300318,0.0007653499,0.0019231454],"category_scores_gemma":[0.009646735,0.00044266594,0.002020749,0.0020897654,0.0015084163,0.00319988,0.0014009018,0.0012939132,0.00023915955],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003161253,0.00031099463,0.0112354355,0.00078148925,0.00036126722,0.0009530721,0.0010881405,0.09472936,0.03413037,0.13252541,0.003974923,0.7195934],"study_design_scores_gemma":[0.00004221079,0.00019958876,0.0038186149,0.00010829215,0.00014301397,0.0010036111,0.0002381193,0.87296313,0.031175952,0.08423059,0.006000642,0.000076267774],"about_ca_topic_score_codex":0.0041382574,"about_ca_topic_score_gemma":0.0035902641,"teacher_disagreement_score":0.004244712,"about_ca_system_score_codex":0.0012375737,"about_ca_system_score_gemma":0.0013359109,"threshold_uncertainty_score":0.010862589},"labels":[],"label_agreement":null},{"id":"W4410344404","doi":"10.1145/3735553","title":"Large Language Models for Automated Web-Form-Test Generation: An Empirical Study","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Test (biology); Software engineering; Web application; Natural language processing; Programming language; World Wide Web; Information retrieval","score_opus":0.10826609347819456,"score_gpt":0.39020246290793753,"score_spread":0.281936369429743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410344404","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9754132,0.0009413905,0.018625537,0.00039686236,0.000042684103,0.0005138186,0.0010537676,0.001078884,0.0019338917],"genre_scores_gemma":[0.96631205,0.00035116333,0.028575733,0.00016178357,0.000034968438,0.0004462553,0.003326062,0.00024307307,0.0005489484],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.97970325,0.013672685,0.0015014908,0.0017710034,0.0028767241,0.00047484922],"domain_scores_gemma":[0.66264033,0.306694,0.007829366,0.012152107,0.008890552,0.0017936482],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02084597,0.0014197938,0.0009199445,0.0018226053,0.00074057933,0.0025519584,0.0020138877,0.001418776,0.002690488],"category_scores_gemma":[0.1549455,0.00079020363,0.0013564496,0.001959577,0.0013777647,0.005385495,0.0019468053,0.0032098119,0.0012510248],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004241603,0.012302609,0.28104794,0.0035507276,0.0006507839,0.0014012964,0.0074223545,0.15701818,0.007697104,0.0049620466,0.016910246,0.5027951],"study_design_scores_gemma":[0.00052588666,0.0021042898,0.06903405,0.0003881955,0.00051731296,0.00068277994,0.0028269272,0.90572435,0.007264513,0.0030777035,0.0076934434,0.00016052775],"about_ca_topic_score_codex":0.009740454,"about_ca_topic_score_gemma":0.0078726495,"teacher_disagreement_score":0.02084597,"about_ca_system_score_codex":0.0023118793,"about_ca_system_score_gemma":0.0021413022,"threshold_uncertainty_score":0.11024529},"labels":[],"label_agreement":null},{"id":"W4410394325","doi":"10.1109/tse.2025.3570897","title":"Using Cooperative Co-Evolutionary Search to Generate Metamorphic Test Cases for Autonomous Driving Systems","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test (biology); Evolutionary algorithm; Artificial intelligence; Distributed computing; Geology","score_opus":0.047251793029374996,"score_gpt":0.3055320633810115,"score_spread":0.25828027035163653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410394325","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5143457,0.0007020927,0.47364137,0.00058379705,0.000092121845,0.00040999995,0.00027206857,0.0030503073,0.006902487],"genre_scores_gemma":[0.8718829,0.00009073766,0.125878,0.00016575055,0.000012936561,0.00020470525,0.0004193887,0.00015064553,0.0011949852],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99882287,0.00045203825,0.000055915923,0.00019125915,0.00033841268,0.00013950033],"domain_scores_gemma":[0.9953235,0.0033510365,0.00027598793,0.00034360585,0.0005356082,0.00017030971],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015311387,0.0012397239,0.00049570814,0.0015082519,0.00036191527,0.0006668704,0.0017683653,0.0011170364,0.0014803839],"category_scores_gemma":[0.0076241754,0.00041328144,0.0006888211,0.0005047108,0.0009592847,0.0007393013,0.0013001015,0.00085570256,0.00021181344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011911885,0.00024575574,0.00815287,0.00011368145,0.000064237065,0.0003885044,0.00013857262,0.8957475,0.007376323,0.0033722366,0.0011252244,0.083156005],"study_design_scores_gemma":[0.000018048104,0.00006983915,0.000291784,0.0000072045814,0.000009374249,0.000030757332,0.000022072756,0.9961909,0.0015291853,0.0014433576,0.0003832427,0.000004238321],"about_ca_topic_score_codex":0.0058220346,"about_ca_topic_score_gemma":0.0075561823,"teacher_disagreement_score":0.0058220346,"about_ca_system_score_codex":0.00082716433,"about_ca_system_score_gemma":0.0013233268,"threshold_uncertainty_score":0.011576295},"labels":[],"label_agreement":null},{"id":"W4410538223","doi":"10.1109/icst62969.2025.10989029","title":"ML-Based Test Case Prioritization: A Research and Production Perspective in CI Environments","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Prioritization; Perspective (graphical); Test (biology); Production (economics); Computer science; Reliability engineering; Risk analysis (engineering); Engineering; Process management; Artificial intelligence; Business; Geology","score_opus":0.04430461546752107,"score_gpt":0.3633840419375879,"score_spread":0.3190794264700668,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410538223","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17172727,0.0022917716,0.7943545,0.009111583,0.0002438393,0.00068309705,0.00074019126,0.014750683,0.0060970527],"genre_scores_gemma":[0.67321473,0.0004595636,0.32308832,0.0005606557,0.00012459543,0.00021946193,0.0010642897,0.00061748986,0.000650964],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9858094,0.007052373,0.0010760535,0.0019636957,0.0032441302,0.00085431145],"domain_scores_gemma":[0.9260203,0.045674704,0.004965107,0.01111987,0.010066757,0.002153208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026839884,0.0020489877,0.0010508988,0.0033827445,0.0007723317,0.0055390242,0.004943778,0.0017531151,0.0015573723],"category_scores_gemma":[0.08950608,0.0007860171,0.0006259084,0.0028922078,0.0022261261,0.008872403,0.0028784056,0.0048872535,0.0006228791],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006845194,0.0014390391,0.05034353,0.0007270304,0.00014816088,0.0004594336,0.0012685223,0.36394745,0.024386333,0.015209245,0.01447887,0.52690786],"study_design_scores_gemma":[0.000078687044,0.00041390132,0.0051601296,0.00012737002,0.000040565734,0.00023900288,0.0006138331,0.9607613,0.014657007,0.012474493,0.0053766347,0.00005697214],"about_ca_topic_score_codex":0.008792123,"about_ca_topic_score_gemma":0.009651452,"teacher_disagreement_score":0.026839884,"about_ca_system_score_codex":0.003086037,"about_ca_system_score_gemma":0.005622246,"threshold_uncertainty_score":0.14194459},"labels":[],"label_agreement":null},{"id":"W4411058752","doi":"10.1145/3727582.3728681","title":"Leveraging LLM Enhanced Commit Messages to Improve Machine Learning Based Test Case Prioritization","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); Ontario Tech University","funders":"","keywords":"Commit; Prioritization; Computer science; Test (biology); Machine learning; Artificial intelligence; Process management; Engineering; Database","score_opus":0.01080975645387326,"score_gpt":0.26644774812461836,"score_spread":0.2556379916707451,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411058752","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27898142,0.00039330753,0.66544384,0.0014060766,0.00022594885,0.0007834781,0.0007297114,0.049389392,0.0026468034],"genre_scores_gemma":[0.6292804,0.00008181018,0.3647056,0.00041005493,0.00007595572,0.00025554118,0.0016477759,0.0012858968,0.0022568502],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99517834,0.0018928383,0.00045147236,0.0007302989,0.0014986749,0.00024823795],"domain_scores_gemma":[0.9577842,0.024328113,0.004812248,0.006046887,0.0060974937,0.00093116803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051420233,0.0012481712,0.00072795054,0.002132284,0.0004130668,0.0015608319,0.0020661976,0.0009075653,0.002207521],"category_scores_gemma":[0.0528041,0.00040211843,0.0005218877,0.00086289836,0.0005949391,0.0022658966,0.0012356514,0.0022510695,0.0009957587],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014719056,0.0015807695,0.029476082,0.00054490153,0.0000974261,0.0007057564,0.0010566266,0.09133901,0.082561,0.004103919,0.008106893,0.7789557],"study_design_scores_gemma":[0.00012395561,0.00059893104,0.00432403,0.000056967696,0.000052741903,0.0002551815,0.00018730952,0.91913474,0.06577509,0.004527331,0.0049092015,0.00005455204],"about_ca_topic_score_codex":0.0031876082,"about_ca_topic_score_gemma":0.005151335,"teacher_disagreement_score":0.0051420233,"about_ca_system_score_codex":0.0010190799,"about_ca_system_score_gemma":0.0024975887,"threshold_uncertainty_score":0.027193904},"labels":[],"label_agreement":null},{"id":"W4411087888","doi":"10.1145/3742894","title":"The Havoc Paradox in Generator-Based Fuzzing","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Fuzz testing; Computer science; Generator (circuit theory); Programming language; Software; Power (physics)","score_opus":0.0603762164017855,"score_gpt":0.3174931715724063,"score_spread":0.25711695517062083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411087888","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43518162,0.0007122447,0.5546521,0.00071311323,0.000049828337,0.00012345928,0.0000872194,0.005007104,0.0034734039],"genre_scores_gemma":[0.9102899,0.000118204465,0.08817828,0.00028046008,0.000012161829,0.000055245197,0.00009142952,0.00036175447,0.00061259663],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9933089,0.0025712051,0.00037492512,0.00096104614,0.0024169215,0.0003669973],"domain_scores_gemma":[0.9726977,0.018838016,0.0017898956,0.0049455184,0.0014013009,0.00032762048],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064800642,0.00070715597,0.0007472973,0.0014283498,0.0006045751,0.0013437523,0.0016957483,0.001208624,0.000853636],"category_scores_gemma":[0.03470755,0.00059378706,0.00065591856,0.0006592985,0.0031288199,0.0033140604,0.0019332258,0.0017837307,0.00017899025],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015841769,0.0005181816,0.03721983,0.00084037485,0.00035787054,0.0016587693,0.0022876614,0.35094917,0.12263222,0.09930183,0.002845074,0.3798048],"study_design_scores_gemma":[0.00012475278,0.00076244975,0.0042147837,0.0001076178,0.00014200051,0.0011857409,0.0002049688,0.8167184,0.100768715,0.071712814,0.003960565,0.00009711801],"about_ca_topic_score_codex":0.0016231107,"about_ca_topic_score_gemma":0.0024276911,"teacher_disagreement_score":0.0064800642,"about_ca_system_score_codex":0.0009357057,"about_ca_system_score_gemma":0.0014237794,"threshold_uncertainty_score":0.034270287},"labels":[],"label_agreement":null},{"id":"W4411113098","doi":"10.18653/v1/2025.findings-naacl.197","title":"TESTEVAL: Benchmarking Large Language Models for Test Case Generation","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada","keywords":"Benchmarking; Computer science; Test (biology); Programming language; Natural language processing; Geology","score_opus":0.032397878695373764,"score_gpt":0.31183753192380914,"score_spread":0.2794396532284354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411113098","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37826398,0.006674871,0.14209993,0.0023068495,0.0007629575,0.0015940032,0.18013135,0.26817343,0.01999263],"genre_scores_gemma":[0.34525427,0.0013582114,0.21307378,0.0010718694,0.00008818088,0.0015838152,0.4237168,0.01112443,0.0027285544],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99286944,0.002688529,0.00089330715,0.0013490536,0.0017750206,0.0004246749],"domain_scores_gemma":[0.97780377,0.014590922,0.0011668414,0.003798082,0.0020868743,0.00055359025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050937175,0.0031082397,0.00077683205,0.005378603,0.0006193518,0.0019067226,0.0045027165,0.0020309812,0.0041490444],"category_scores_gemma":[0.033505,0.0009804968,0.0021747593,0.0047314423,0.0011249,0.0034117135,0.002104221,0.0020114684,0.0028235002],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022773517,0.0026763745,0.06028756,0.0073710578,0.0016225224,0.0009197936,0.0009187815,0.26100218,0.013600847,0.011709063,0.36756927,0.2700452],"study_design_scores_gemma":[0.0011751197,0.00094636955,0.015504866,0.000472598,0.00030122953,0.00070342567,0.00043745746,0.85617566,0.02486428,0.01461195,0.084628984,0.00017809955],"about_ca_topic_score_codex":0.0130375745,"about_ca_topic_score_gemma":0.022194855,"teacher_disagreement_score":0.0130375745,"about_ca_system_score_codex":0.0020658583,"about_ca_system_score_gemma":0.0026957046,"threshold_uncertainty_score":0.026938498},"labels":[],"label_agreement":null},{"id":"W4411271577","doi":"10.1109/msr66628.2025.00022","title":"Revisiting Defects4J for Fault Localization in Diverse Development Scenarios","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Concordia University","funders":"","keywords":"Computer science; Fault (geology); Development (topology); Systems engineering; Geology; Engineering; Seismology; Mathematics","score_opus":0.023507470805200638,"score_gpt":0.2886050368149066,"score_spread":0.26509756600970596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411271577","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6486614,0.008144978,0.051186185,0.0028492159,0.0013241571,0.0008171786,0.20236984,0.07290682,0.011740318],"genre_scores_gemma":[0.39792526,0.0010258191,0.097663336,0.00094389846,0.0002124049,0.0007410302,0.49291947,0.005934549,0.002634175],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9840941,0.0036648933,0.0020577037,0.0034659898,0.005708549,0.001008813],"domain_scores_gemma":[0.94541436,0.024398046,0.00661757,0.011197196,0.01031003,0.0020628222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0098925885,0.0023989037,0.0011993141,0.0097625,0.0012573639,0.003745684,0.003955576,0.0020463208,0.0017984699],"category_scores_gemma":[0.05308875,0.0008245913,0.0018867286,0.0066274335,0.0012942047,0.0038743196,0.003255447,0.0027979827,0.0018995645],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025475514,0.0021487055,0.2708367,0.0065458897,0.0017286292,0.0014749644,0.0019051969,0.056755405,0.030676614,0.007852294,0.36321208,0.25431597],"study_design_scores_gemma":[0.001806299,0.0036211228,0.3416996,0.0014333657,0.0009575321,0.003084276,0.0021390538,0.3083065,0.03988186,0.01955723,0.27683854,0.00067455985],"about_ca_topic_score_codex":0.01833352,"about_ca_topic_score_gemma":0.029657133,"teacher_disagreement_score":0.01833352,"about_ca_system_score_codex":0.0020332576,"about_ca_system_score_gemma":0.0030447068,"threshold_uncertainty_score":0.05231762},"labels":[],"label_agreement":null},{"id":"W4411300576","doi":"10.1007/978-3-031-95497-9_2","title":"Temporal and Spatial Fault Detection for Connected Cyber-Physical Systems","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Cyber-physical system; Fault (geology); Fault detection and isolation; Real-time computing; Artificial intelligence; Operating system; Seismology; Geology","score_opus":0.01582217726198094,"score_gpt":0.25824702030820834,"score_spread":0.2424248430462274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411300576","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047814384,0.0008908673,0.9447133,0.00015997692,0.00012931306,0.000024605588,0.00008727674,0.0008208088,0.005359514],"genre_scores_gemma":[0.84775734,0.00080419116,0.14409958,0.000060360973,0.00007784644,0.000031185577,0.00014616616,0.0000745719,0.0069488403],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997185,0.00003797091,0.000016422624,0.00006216675,0.00013412072,0.00003078317],"domain_scores_gemma":[0.99899656,0.00058503554,0.000100263285,0.00011388753,0.00016974403,0.00003456392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002833658,0.00053788535,0.00040132474,0.0009712777,0.00023774405,0.0006148984,0.00069464237,0.00043683342,0.0029721486],"category_scores_gemma":[0.0022234693,0.00020634121,0.0003887502,0.000663439,0.0005467015,0.0010630672,0.0006159645,0.00047901334,0.0002614998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006880462,0.00009901792,0.002661488,0.00041243463,0.00008014747,0.00038438366,0.00018958148,0.2959975,0.035796482,0.077199124,0.0034992958,0.58299255],"study_design_scores_gemma":[0.000010556807,0.00013337705,0.0010698568,0.000027791155,0.000029687055,0.00029429563,0.000036441008,0.9436175,0.009412365,0.043225825,0.0021309985,0.000011219832],"about_ca_topic_score_codex":0.0014269745,"about_ca_topic_score_gemma":0.0021001636,"teacher_disagreement_score":0.0029721486,"about_ca_system_score_codex":0.00037270342,"about_ca_system_score_gemma":0.0003743644,"threshold_uncertainty_score":0.00994283},"labels":[],"label_agreement":null},{"id":"W4411440374","doi":"10.1007/s10664-025-10661-x","title":"Supporting multi-dimensional unit test classification","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Unit testing; Test (biology); Unit (ring theory); Artificial intelligence; Pattern recognition (psychology); Geology; Psychology; Operating system; Mathematics education","score_opus":0.04423106979949689,"score_gpt":0.34067254859068613,"score_spread":0.29644147879118926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411440374","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27429515,0.001407177,0.68895686,0.0008348183,0.00017895999,0.00057920144,0.008179392,0.022470292,0.0030980934],"genre_scores_gemma":[0.5206039,0.0002100652,0.45826927,0.00029761464,0.00008332834,0.00033518326,0.018589443,0.0005689745,0.0010422777],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9909928,0.0014498674,0.0015204047,0.0022501606,0.003304785,0.00048198632],"domain_scores_gemma":[0.95365673,0.022926288,0.005153484,0.008529459,0.008679735,0.001054351],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059470935,0.0014203084,0.0020008322,0.006860353,0.0009260309,0.0031925468,0.0034195762,0.0020325785,0.0012118662],"category_scores_gemma":[0.040013388,0.00045919864,0.0014972413,0.005032581,0.000911596,0.0035971196,0.0026557534,0.0021468091,0.0011141459],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007978002,0.00096463197,0.14386603,0.0007458571,0.000261658,0.00076711003,0.00088915986,0.03793439,0.019581351,0.0043029436,0.014727284,0.77516174],"study_design_scores_gemma":[0.00006356014,0.00018190473,0.018067634,0.000100968784,0.00008555252,0.00059177063,0.00038503736,0.9374019,0.023374945,0.013646349,0.0060276985,0.00007269984],"about_ca_topic_score_codex":0.0058789174,"about_ca_topic_score_gemma":0.01080994,"teacher_disagreement_score":0.006860353,"about_ca_system_score_codex":0.0012497718,"about_ca_system_score_gemma":0.0019212636,"threshold_uncertainty_score":0.031451643},"labels":[],"label_agreement":null},{"id":"W4411450040","doi":"10.1145/3715741","title":"Understanding and Characterizing Mock Assertions in Unit Tests","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Assertion; Test (biology); Testability; Complement (music); Unit testing; Programming language; Control flow; Test case; Software engineering; Reliability engineering; Machine learning; Software","score_opus":0.05208794963468529,"score_gpt":0.26277533769608935,"score_spread":0.21068738806140405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411450040","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16103145,0.0007137136,0.8311719,0.00054147386,0.00006836399,0.00031042792,0.0002785719,0.0030717538,0.00281233],"genre_scores_gemma":[0.7063916,0.00031969798,0.29005092,0.0003338907,0.0000673085,0.00043195055,0.0006862753,0.0007155598,0.0010028699],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9722415,0.012721437,0.0027931128,0.002828043,0.008027589,0.0013882483],"domain_scores_gemma":[0.71898556,0.19552024,0.029523859,0.03528652,0.01867695,0.0020069538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015965616,0.0011710208,0.00083554396,0.004196637,0.0009501083,0.0046691876,0.0027490049,0.0027275132,0.0013389476],"category_scores_gemma":[0.17984629,0.0013761406,0.0010335029,0.001986195,0.005152537,0.012391835,0.0034992145,0.0020848766,0.00051997474],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001001232,0.0006571236,0.15441646,0.0018502233,0.0002765476,0.00582518,0.01834038,0.14265609,0.040878158,0.3354688,0.004641004,0.29398882],"study_design_scores_gemma":[0.00010663513,0.00075552345,0.015785739,0.0012781398,0.00027138396,0.003743777,0.0026332967,0.5179235,0.053189196,0.3707064,0.03330152,0.00030478725],"about_ca_topic_score_codex":0.0030503094,"about_ca_topic_score_gemma":0.002743011,"teacher_disagreement_score":0.015965616,"about_ca_system_score_codex":0.0015584194,"about_ca_system_score_gemma":0.002327923,"threshold_uncertainty_score":0.084435225},"labels":[],"label_agreement":null},{"id":"W4411523015","doi":"10.1145/3728876","title":"MoDitector: Module-Directed Testing for Autonomous Driving Systems","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Root cause; Computer science; Debugging; Reliability (semiconductor); Reliability engineering; Root cause analysis; Process (computing); Scenario testing; Embedded system; Engineering; Artificial intelligence","score_opus":0.016940612286937884,"score_gpt":0.23593435620730108,"score_spread":0.2189937439203632,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411523015","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18658833,0.0006387783,0.73604816,0.00037834977,0.00014233353,0.00067971175,0.0009830919,0.06771174,0.006829457],"genre_scores_gemma":[0.70559293,0.00018296816,0.28782985,0.00028091646,0.000025238036,0.00034923333,0.0012271835,0.0019197102,0.0025920623],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99887186,0.00029428606,0.00006594134,0.00018593624,0.000471233,0.00011069236],"domain_scores_gemma":[0.9969541,0.0017778024,0.00030649416,0.0004991258,0.00036630774,0.000096147785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001273779,0.001298076,0.00033401162,0.0009163119,0.00025423104,0.0005513009,0.002340639,0.0009545211,0.003346474],"category_scores_gemma":[0.005496521,0.00043549106,0.0006335419,0.00028773415,0.00083833875,0.001353927,0.0009855236,0.00094404275,0.0005986017],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086137943,0.00049857004,0.02393115,0.0010338032,0.00022341085,0.0015035381,0.0008412727,0.33204192,0.16534922,0.013717144,0.016129771,0.4438688],"study_design_scores_gemma":[0.00012396037,0.0007011437,0.0030369745,0.0000635082,0.00006045945,0.0006703407,0.00007748,0.85589176,0.119308576,0.0066024144,0.013395785,0.00006758596],"about_ca_topic_score_codex":0.0024848378,"about_ca_topic_score_gemma":0.0031547588,"teacher_disagreement_score":0.003346474,"about_ca_system_score_codex":0.0005487496,"about_ca_system_score_gemma":0.00085482275,"threshold_uncertainty_score":0.011195064},"labels":[],"label_agreement":null},{"id":"W4411551806","doi":"10.1109/icse55347.2025.00141","title":"Feature-Driven End-to-End Test Generation","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"End-to-end principle; Computer science; Test (biology); Feature (linguistics); Artificial intelligence; Geology","score_opus":0.020466836786324962,"score_gpt":0.27702694174716963,"score_spread":0.2565601049608447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411551806","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13251649,0.00034359613,0.8040942,0.00049731,0.0001269695,0.0007617167,0.002292083,0.054374013,0.00499366],"genre_scores_gemma":[0.4766448,0.00018905541,0.50664985,0.00045897093,0.00003764297,0.0008865075,0.00828789,0.0040949048,0.0027503546],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99643695,0.0011522872,0.0002557518,0.00056950114,0.0012933837,0.00029219166],"domain_scores_gemma":[0.98484105,0.008543209,0.00085375557,0.0024750833,0.0028975564,0.00038933626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024535812,0.0015032064,0.0006295375,0.001873883,0.00031218544,0.0010408979,0.0024233782,0.0011979464,0.0037289879],"category_scores_gemma":[0.020507913,0.00053743157,0.0009827206,0.0008529376,0.0006720114,0.0013592067,0.0016486486,0.0011993136,0.0016498121],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009361898,0.0013256758,0.023709109,0.0009257607,0.00025337932,0.002590376,0.0006291416,0.17247836,0.09230539,0.012673201,0.035079874,0.6570935],"study_design_scores_gemma":[0.00022505627,0.0004642255,0.003315279,0.0000783808,0.000059933187,0.00087721366,0.00013523964,0.8769232,0.09681966,0.009610995,0.011431742,0.00005905462],"about_ca_topic_score_codex":0.0019179992,"about_ca_topic_score_gemma":0.002730194,"teacher_disagreement_score":0.0037289879,"about_ca_system_score_codex":0.000672764,"about_ca_system_score_gemma":0.0016234764,"threshold_uncertainty_score":0.012975931},"labels":[],"label_agreement":null},{"id":"W4411799800","doi":"10.1109/se4ads66461.2025.00007","title":"AI-Augmented Metamorphic Testing for Comprehensive Validation of Autonomous Vehicles","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Metamorphic rock; Geology","score_opus":0.061548962206652055,"score_gpt":0.3180466203913648,"score_spread":0.2564976581847127,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411799800","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08908694,0.00028009174,0.8968482,0.00037824415,0.00007593035,0.00016542363,0.0001631517,0.00803409,0.004967873],"genre_scores_gemma":[0.6632369,0.00017959521,0.3335215,0.00024057859,0.000027106755,0.00014818706,0.00037870495,0.0005542404,0.001713119],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996274,0.0017305991,0.00024332327,0.0003794826,0.0011803417,0.00019228602],"domain_scores_gemma":[0.9890315,0.005573091,0.00094445335,0.00270308,0.0015131427,0.00023468187],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003362075,0.0006348505,0.00033251615,0.0010119785,0.00033242835,0.0011257811,0.0015938636,0.0009249154,0.0022794453],"category_scores_gemma":[0.013747112,0.00036548843,0.00064739765,0.00039690686,0.0016549622,0.0017406854,0.0019159629,0.0012459642,0.00032761053],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047959574,0.00048794798,0.015199679,0.000729571,0.00014394254,0.0008636857,0.0011820223,0.38842723,0.15044,0.08743722,0.0049982863,0.3496108],"study_design_scores_gemma":[0.00003931624,0.0002203565,0.0013834323,0.00007802439,0.000024061515,0.00027117057,0.00008122442,0.9237914,0.044791542,0.022753427,0.006536473,0.000029578956],"about_ca_topic_score_codex":0.0017695827,"about_ca_topic_score_gemma":0.0024227225,"teacher_disagreement_score":0.003362075,"about_ca_system_score_codex":0.0007093151,"about_ca_system_score_gemma":0.0010754563,"threshold_uncertainty_score":0.017780542},"labels":[],"label_agreement":null},{"id":"W4411950604","doi":"10.1109/forge66646.2025.00020","title":"Testing Refactoring Engine via Historical Bug Report driven LLM","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code refactoring; Computer science; Programming language; Software engineering; Software","score_opus":0.02902503802755674,"score_gpt":0.2717983015721504,"score_spread":0.24277326354459366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411950604","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5764394,0.00070431625,0.3813134,0.000243674,0.00007916552,0.00046764777,0.001056774,0.037259467,0.0024360695],"genre_scores_gemma":[0.79372597,0.0001467287,0.20245656,0.00011041082,0.000017752489,0.00019554584,0.0014500149,0.0007633671,0.0011336121],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99494743,0.0010713342,0.00038275754,0.0014438805,0.0019124405,0.00024218745],"domain_scores_gemma":[0.9799534,0.006651678,0.0031561425,0.005285995,0.0045008063,0.00045194887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039245547,0.0010728479,0.00045237684,0.002202498,0.0002521098,0.00068089657,0.0019645025,0.00073646085,0.0009749802],"category_scores_gemma":[0.019818507,0.00045660528,0.0005784632,0.00075197,0.0005721985,0.0013935413,0.0009127781,0.0008289424,0.00052780606],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008872128,0.0012357145,0.10477287,0.0008691701,0.00022231696,0.0017492939,0.0016545231,0.06830437,0.16068892,0.0030755247,0.004452828,0.6520873],"study_design_scores_gemma":[0.0002277487,0.0024831633,0.045357622,0.00013646904,0.0002964251,0.0013880696,0.00031419608,0.6970916,0.23809797,0.0027011738,0.011721313,0.00018420501],"about_ca_topic_score_codex":0.0022901392,"about_ca_topic_score_gemma":0.0029306433,"teacher_disagreement_score":0.0039245547,"about_ca_system_score_codex":0.0006127251,"about_ca_system_score_gemma":0.0009096525,"threshold_uncertainty_score":0.020755231},"labels":[],"label_agreement":null},{"id":"W4412537309","doi":"10.1109/ast66626.2025.00009","title":"Simulink Mutation Testing using CodeBERT","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Mutation; Mutation testing; Software testing; Genetics; Programming language; Biology; Software","score_opus":0.049422659587397964,"score_gpt":0.32144220324181105,"score_spread":0.2720195436544131,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412537309","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08514984,0.0002360435,0.8575764,0.00022269963,0.00009540622,0.00012860993,0.00044072667,0.048864562,0.0072857416],"genre_scores_gemma":[0.6002081,0.0001756791,0.39074275,0.00016510263,0.000019176041,0.0001332435,0.00088362704,0.0037333788,0.0039389976],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.99865204,0.00032231893,0.0000703358,0.0001974683,0.0006548098,0.00010306591],"domain_scores_gemma":[0.99564725,0.002417265,0.00048332688,0.00079794053,0.00056359015,0.00009064916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011066645,0.0012158765,0.0004275336,0.0010459801,0.00032746035,0.000677673,0.0015536626,0.00086014875,0.0029166013],"category_scores_gemma":[0.007146557,0.00042744947,0.0005719332,0.0003549698,0.0009511128,0.0010833312,0.0009329919,0.0008070205,0.0006507033],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008210846,0.00034697316,0.009438277,0.0006182408,0.00018755456,0.0014419742,0.0005670478,0.6096101,0.10115437,0.05202948,0.009511767,0.21427311],"study_design_scores_gemma":[0.00004914224,0.00018656516,0.00055165257,0.000059665705,0.000035992478,0.0004109053,0.00003601246,0.89832497,0.08168909,0.007097269,0.011520817,0.00003791997],"about_ca_topic_score_codex":0.0035007293,"about_ca_topic_score_gemma":0.0043918965,"teacher_disagreement_score":0.0035007293,"about_ca_system_score_codex":0.000745105,"about_ca_system_score_gemma":0.001189361,"threshold_uncertainty_score":0.009757042},"labels":[],"label_agreement":null},{"id":"W4412703957","doi":"10.1145/3696630.3728608","title":"A Tool for Generating Exceptional Behavior Tests With Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Alliance de recherche numérique du Canada; University of Texas at Austin; Cisco Systems; National Science Foundation","keywords":"Computer science; Programming language; Artificial intelligence; Natural language processing","score_opus":0.03533501993292639,"score_gpt":0.31782319660617253,"score_spread":0.28248817667324616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412703957","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070993523,0.00015007936,0.7154246,0.00031197924,0.0000941939,0.00033134694,0.0044670193,0.26792055,0.004200891],"genre_scores_gemma":[0.124522015,0.00032573284,0.7956105,0.00047025853,0.00006350029,0.0012781636,0.018038401,0.052062314,0.0076291487],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975484,0.00072696235,0.00028489847,0.0004355437,0.0008552209,0.0001488791],"domain_scores_gemma":[0.98857087,0.007940791,0.0006624224,0.001686156,0.00091827224,0.00022153334],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035738016,0.002428494,0.00070537697,0.0022746376,0.0007511177,0.0021837023,0.0026820595,0.0017128461,0.020363722],"category_scores_gemma":[0.020794947,0.002148898,0.0020263088,0.0008745727,0.001270652,0.0044935397,0.0039927806,0.002959709,0.0069547957],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010238414,0.00095774786,0.019099163,0.003190887,0.00044225345,0.0033411947,0.0032405998,0.09731524,0.05694107,0.094391264,0.25687355,0.46318325],"study_design_scores_gemma":[0.00040047322,0.00026072524,0.0017986431,0.00051698094,0.000117829615,0.0015907021,0.0003434123,0.6649532,0.059953574,0.067440405,0.20241162,0.00021244552],"about_ca_topic_score_codex":0.0025171393,"about_ca_topic_score_gemma":0.004676433,"teacher_disagreement_score":0.020363722,"about_ca_system_score_codex":0.00091444544,"about_ca_system_score_gemma":0.0020814994,"threshold_uncertainty_score":0.0681234},"labels":[],"label_agreement":null},{"id":"W4413185076","doi":"10.1016/j.neucom.2025.131208","title":"A distance for mixed-variable and hierarchical domains with meta variables","year":2025,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Trottier Family Foundation; Fonds de recherche du Québec – Nature et technologies; European Commission; Fonds de recherche du Québec; Natural Sciences and Engineering Research Council of Canada; Office National d'études et de Recherches Aérospatiales; HORIZON EUROPE Framework Programme; Agence Nationale de la Recherche","keywords":"Variable (mathematics); Computer science; Mathematics; Artificial intelligence; Statistics","score_opus":0.015196723601141998,"score_gpt":0.2552330632174972,"score_spread":0.24003633961635523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413185076","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024916904,0.0002374629,0.9719212,0.00029929794,0.000046633635,0.000039701597,0.00011303258,0.00020141706,0.0022243473],"genre_scores_gemma":[0.3132268,0.00019603422,0.68085337,0.00017843841,0.00008763264,0.00017099352,0.000466473,0.00019538209,0.004624895],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99443215,0.0026980944,0.00039898927,0.0009647557,0.0011883668,0.00031759913],"domain_scores_gemma":[0.9788889,0.014189291,0.00094707153,0.0026517871,0.0022601977,0.0010627373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055462206,0.00075443834,0.0011581354,0.0019051045,0.0010963086,0.0018165678,0.003104966,0.0019006332,0.0026866116],"category_scores_gemma":[0.028013134,0.00045514296,0.0012149157,0.0015093646,0.0020426647,0.006411748,0.0056586447,0.004043378,0.00064765627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005927655,0.00033155107,0.0044186264,0.0003299761,0.00011168734,0.0001306151,0.00056157436,0.081703156,0.0034907158,0.62301546,0.0050406214,0.2802732],"study_design_scores_gemma":[0.00004293768,0.00036222613,0.0009845585,0.0000703287,0.000034844765,0.0001269494,0.00016545501,0.45673296,0.0025150862,0.532422,0.0065011983,0.000041535746],"about_ca_topic_score_codex":0.001196167,"about_ca_topic_score_gemma":0.0016767947,"teacher_disagreement_score":0.0055462206,"about_ca_system_score_codex":0.0014116633,"about_ca_system_score_gemma":0.0015207777,"threshold_uncertainty_score":0.029331625},"labels":[],"label_agreement":null},{"id":"W4413212461","doi":"10.1109/ms.2025.3597574","title":"When AI-Generated Unit Tests Validate Bugs: The Risk of Faulty Assertions","year":2025,"lang":"en","type":"article","venue":"IEEE Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Software bug; Computer science; Unit testing; Software engineering; Programming language; Reliability engineering; Software; Engineering","score_opus":0.025979699802465107,"score_gpt":0.2993781457658547,"score_spread":0.2733984459633896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413212461","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5405851,0.0020220308,0.42500594,0.005361979,0.00044403912,0.00027983237,0.00044875132,0.015215508,0.010636806],"genre_scores_gemma":[0.9229412,0.00019267983,0.073488116,0.0008663571,0.00006825737,0.00006793318,0.0002854633,0.0010800275,0.0010100141],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9740121,0.011609657,0.0010401466,0.0020237148,0.010535578,0.0007788102],"domain_scores_gemma":[0.64866924,0.26903588,0.02481758,0.040514022,0.015635662,0.001327582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017964972,0.00065370806,0.00058227783,0.0017856098,0.00051281584,0.0023079596,0.0020395613,0.002451472,0.0023565511],"category_scores_gemma":[0.22607774,0.00056892796,0.000561498,0.0007805277,0.0022642952,0.0038731752,0.0019064585,0.0016243057,0.0010282682],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022089558,0.0005613162,0.23362897,0.0013286899,0.0004990993,0.00650636,0.008270966,0.13527179,0.049369372,0.072219886,0.014053363,0.47608116],"study_design_scores_gemma":[0.00035656156,0.0022871809,0.02792418,0.0015093515,0.00044404974,0.008344737,0.0020395678,0.699391,0.12967794,0.09815353,0.029598115,0.00027377834],"about_ca_topic_score_codex":0.0018578591,"about_ca_topic_score_gemma":0.0013138978,"teacher_disagreement_score":0.017964972,"about_ca_system_score_codex":0.00096260494,"about_ca_system_score_gemma":0.0011606629,"threshold_uncertainty_score":0.09500903},"labels":[],"label_agreement":null},{"id":"W4413373092","doi":"10.1007/978-3-032-02018-5_35","title":"Uncovering Unsafe Feature Interactions in Vehicle Control Using Generative AI and Digital Twins","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Critical Systems Labs","funders":"","keywords":"Computer science; Generative grammar; Feature (linguistics); Control (management); Human–computer interaction; Artificial intelligence","score_opus":0.017737517385578803,"score_gpt":0.27524651233036507,"score_spread":0.25750899494478624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413373092","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32037416,0.00045308113,0.66552943,0.0003244157,0.00004220843,0.000061202765,0.0001853063,0.0018609617,0.011169141],"genre_scores_gemma":[0.8803515,0.00012228402,0.11642249,0.00005045645,0.000008612067,0.000024858318,0.00016128889,0.00017924534,0.0026792325],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947387,0.00011615889,0.000018790624,0.00012785291,0.00019534826,0.000067939414],"domain_scores_gemma":[0.9974558,0.0017168049,0.00018383986,0.0004364296,0.00015396577,0.000053146206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000483178,0.000376544,0.0003720561,0.0012123672,0.0005774799,0.0010495766,0.001025805,0.00069710606,0.002131661],"category_scores_gemma":[0.003599881,0.00044882423,0.00050968677,0.0008964425,0.0014100879,0.0016232653,0.0013927434,0.0012416709,0.00029102276],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003395356,0.00028341205,0.025667189,0.00037289888,0.00013574534,0.0013102054,0.0034847034,0.16231284,0.065206304,0.2156757,0.002491902,0.52271956],"study_design_scores_gemma":[0.000014962491,0.00009180771,0.0060064024,0.000049480328,0.000044906803,0.00041141518,0.0007713259,0.8518306,0.019913625,0.11536326,0.005465106,0.000037146343],"about_ca_topic_score_codex":0.003566323,"about_ca_topic_score_gemma":0.0051577343,"teacher_disagreement_score":0.003566323,"about_ca_system_score_codex":0.000525792,"about_ca_system_score_gemma":0.00054624444,"threshold_uncertainty_score":0.0071311593},"labels":[],"label_agreement":null},{"id":"W4413387697","doi":"10.1016/j.infsof.2025.107874","title":"Assessing the impact of tuning parameter in instance selection based bug resolution classification","year":2025,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Selection (genetic algorithm); Resolution (logic); Computer science; Artificial intelligence; Machine learning; Data mining","score_opus":0.02271485433639512,"score_gpt":0.3239449150073322,"score_spread":0.30123006067093705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413387697","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97530633,0.0013712778,0.017057588,0.0002690409,0.00015978972,0.00009295429,0.0005626514,0.0039046959,0.0012756294],"genre_scores_gemma":[0.9790055,0.00011476314,0.019473385,0.000087924,0.000029237106,0.000024666975,0.0007064896,0.00012521668,0.00043272018],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99432915,0.0018960875,0.0006236476,0.0013443581,0.0014051101,0.00040167608],"domain_scores_gemma":[0.9341966,0.04888158,0.004033043,0.006819225,0.0046462,0.001423277],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046187113,0.0010245709,0.0010038753,0.0020903323,0.0003962364,0.0014000682,0.0014019315,0.0015081428,0.00071655394],"category_scores_gemma":[0.044030383,0.00028499632,0.0006965685,0.0013648135,0.00039356714,0.0015995214,0.0005628025,0.001233395,0.00032272527],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0066420357,0.0027012716,0.28536683,0.0005963955,0.00095027837,0.00047038263,0.00035771547,0.11422812,0.052107494,0.0006492586,0.0058363997,0.5300938],"study_design_scores_gemma":[0.00032283858,0.0028204478,0.09883481,0.00009373551,0.00077219045,0.0006076851,0.0003424033,0.8555661,0.037908897,0.0009543026,0.0016705756,0.00010609476],"about_ca_topic_score_codex":0.0046634977,"about_ca_topic_score_gemma":0.003939937,"teacher_disagreement_score":0.0046634977,"about_ca_system_score_codex":0.0005077002,"about_ca_system_score_gemma":0.0010300841,"threshold_uncertainty_score":0.024426341},"labels":[],"label_agreement":null},{"id":"W4413863275","doi":"10.1007/978-3-031-94533-5_21","title":"Model-Based Testing of Non-deterministic Systems","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science","score_opus":0.03303807198923141,"score_gpt":0.27285629151377133,"score_spread":0.2398182195245399,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413863275","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03140515,0.0013052776,0.9571585,0.00029297202,0.00010034358,0.00003890537,0.000072283874,0.0014827932,0.008143709],"genre_scores_gemma":[0.7730543,0.0008097548,0.21572381,0.0001261048,0.00006365863,0.00010619819,0.00025856082,0.0003705451,0.009487022],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985267,0.00051388,0.00005327496,0.0001542933,0.00065842445,0.00009338934],"domain_scores_gemma":[0.9963966,0.0026854873,0.00013951377,0.00051809405,0.00020766445,0.000052646694],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000921884,0.00073905656,0.000708392,0.00058415584,0.00024158125,0.00087442144,0.0015010936,0.0009173043,0.0026298922],"category_scores_gemma":[0.0049958783,0.00063226523,0.000725051,0.00047995374,0.00090831565,0.001627962,0.0008144877,0.0014037711,0.00039595604],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003560167,0.00018671983,0.0013354748,0.00038809603,0.000102901824,0.00033112237,0.00020183431,0.5237859,0.026237763,0.20757076,0.005595799,0.23390757],"study_design_scores_gemma":[0.000026372401,0.00008249243,0.00032188196,0.0000393458,0.000021388003,0.00015161312,0.000010748534,0.8708586,0.00862147,0.11639278,0.003461044,0.000012240632],"about_ca_topic_score_codex":0.0013570014,"about_ca_topic_score_gemma":0.0020720856,"teacher_disagreement_score":0.0026298922,"about_ca_system_score_codex":0.0008391128,"about_ca_system_score_gemma":0.00059979275,"threshold_uncertainty_score":0.008797884},"labels":[],"label_agreement":null},{"id":"W4414093251","doi":"10.3390/make7030097","title":"A Review of Large Language Models for Automated Test Case Generation","year":2025,"lang":"en","type":"review","venue":"Machine Learning and Knowledge Extraction","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Test (biology); Natural language generation; Natural language; Focus (optics); Software; Test case","score_opus":0.04220589820370538,"score_gpt":0.4041485293505684,"score_spread":0.361942631146863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414093251","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00034392145,0.99131846,0.005605451,0.00039829532,0.00013996143,0.00011854121,0.00022561126,0.00010980053,0.0017399681],"genre_scores_gemma":[0.0035134288,0.9855099,0.009345259,0.00034387107,0.000090933594,0.00023533645,0.00048289468,0.00003908633,0.00043923577],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9967181,0.001130211,0.0007648747,0.00032122326,0.0009795275,0.00008604234],"domain_scores_gemma":[0.9754411,0.020293072,0.0013368505,0.0005224252,0.002246443,0.00016009308],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005450453,0.0016742153,0.0020156233,0.009180949,0.00043118303,0.0017293008,0.0025860686,0.0015119043,0.005492438],"category_scores_gemma":[0.023244804,0.0009896924,0.0025464564,0.008276055,0.0006627603,0.0027701173,0.0011512729,0.001370236,0.0023476153],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007015371,0.00007878715,0.0003899224,0.09907699,0.00027035305,0.00014644524,0.0001924858,0.0019673593,0.0007962887,0.004041777,0.014465775,0.8785036],"study_design_scores_gemma":[0.00006334857,0.00036333612,0.002806632,0.1358625,0.0022642065,0.0015474217,0.00026462437,0.002977339,0.0023476894,0.007150427,0.8442238,0.00012869928],"about_ca_topic_score_codex":0.0048487633,"about_ca_topic_score_gemma":0.0073524527,"teacher_disagreement_score":0.009180949,"about_ca_system_score_codex":0.001642625,"about_ca_system_score_gemma":0.0059868223,"threshold_uncertainty_score":0.028825104},"labels":[],"label_agreement":null},{"id":"W4414193951","doi":"10.1007/978-3-032-05188-2_10","title":"Test Generation for Deep Reinforcement Learning Using LRP-Guided Mutation of Classified Configurations","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec en Outaouais","funders":"","keywords":"Reinforcement learning; Classifier (UML); Relevance (law); Key (lock); Binary number; Binary classification","score_opus":0.057442439023788226,"score_gpt":0.3055683181089162,"score_spread":0.24812587908512798,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414193951","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13002586,0.00030105736,0.85610664,0.00037822698,0.000109306755,0.00019888848,0.00022975757,0.006545053,0.0061052465],"genre_scores_gemma":[0.8254335,0.000041776584,0.1708784,0.00017747236,0.000017905655,0.00018712167,0.00026204664,0.00033533233,0.0026664266],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992623,0.00022303792,0.000037376318,0.00017678954,0.00016463173,0.0001358127],"domain_scores_gemma":[0.9971155,0.0019190486,0.00017024267,0.00027022534,0.0004002987,0.00012466616],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011901293,0.0009425522,0.00090402673,0.0007284037,0.0003150306,0.00058549613,0.0020452256,0.0012958278,0.0063883695],"category_scores_gemma":[0.0050015063,0.00046916254,0.0005522712,0.00031345861,0.0009100756,0.00084884034,0.0012376108,0.0014843112,0.0007564638],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047823056,0.00034612807,0.0024367787,0.00015864013,0.00005647336,0.00024953505,0.0001146058,0.6515449,0.01206855,0.006770854,0.004202129,0.32157308],"study_design_scores_gemma":[0.000017456778,0.000033755754,0.000100453944,0.0000074270806,0.0000046091213,0.000018089238,0.000004898422,0.99631554,0.0015107194,0.001844572,0.00013887534,0.00000364792],"about_ca_topic_score_codex":0.00512822,"about_ca_topic_score_gemma":0.005216289,"teacher_disagreement_score":0.0063883695,"about_ca_system_score_codex":0.0012850949,"about_ca_system_score_gemma":0.0013380698,"threshold_uncertainty_score":0.021371186},"labels":[],"label_agreement":null},{"id":"W4414276204","doi":"10.5753/sast.2025.13826","title":"Detecção de Conflitos Semânticos com Testes Gerados por LLM","year":2025,"lang":"pt","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Merge (version control)","score_opus":0.023181220685524074,"score_gpt":0.3002464998347447,"score_spread":0.27706527914922063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414276204","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13566962,0.0038957258,0.7648528,0.00322181,0.00090821524,0.00085431134,0.0028256255,0.057697587,0.03007434],"genre_scores_gemma":[0.4519054,0.0012889322,0.5046812,0.0014928195,0.00019305009,0.0007423081,0.0047766534,0.008861135,0.026058368],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99135673,0.0013658317,0.00084409193,0.0014841297,0.0042048595,0.0007444952],"domain_scores_gemma":[0.9783226,0.006733987,0.0014440982,0.007398892,0.0053525274,0.0007479218],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067452295,0.001645886,0.0012578723,0.0029776532,0.0022425042,0.0061640767,0.0029965234,0.002171589,0.007739704],"category_scores_gemma":[0.030293768,0.0015247929,0.0020361333,0.002547754,0.0028341503,0.011973186,0.007585152,0.0036916025,0.003268827],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019856507,0.0004747408,0.040928762,0.0034478807,0.00046993425,0.0023584403,0.012076366,0.022600932,0.084377795,0.12553354,0.045823645,0.65992224],"study_design_scores_gemma":[0.00020809563,0.00060316204,0.01199586,0.0012533723,0.0005700045,0.0019125118,0.005762673,0.17303537,0.24728401,0.1248709,0.43217003,0.00033404966],"about_ca_topic_score_codex":0.008601972,"about_ca_topic_score_gemma":0.011947355,"teacher_disagreement_score":0.008601972,"about_ca_system_score_codex":0.002893411,"about_ca_system_score_gemma":0.004756599,"threshold_uncertainty_score":0.035672605},"labels":[],"label_agreement":null},{"id":"W4414404854","doi":"10.1109/tse.2025.3612253","title":"MetaSel: A Test Selection Approach for Fine-Tuned DNN Models","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Ottawa","funders":"Alliance de recherche numérique du Canada; Mitacs; Huawei Technologies","keywords":"Covariate; Software deployment; Model selection; Selection (genetic algorithm); Context (archaeology); Subspace topology; Statistical hypothesis testing; Test data; Probability distribution","score_opus":0.01927186097653321,"score_gpt":0.24144192629660027,"score_spread":0.22217006532006706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414404854","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10118463,0.0024629848,0.8488248,0.0005733983,0.00024944465,0.0005115588,0.0016282456,0.040826585,0.0037383656],"genre_scores_gemma":[0.55665493,0.00042849724,0.42513007,0.001576354,0.00016937207,0.0007668921,0.007639664,0.003337156,0.0042970343],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969812,0.001019557,0.0002531388,0.0009037023,0.0006214065,0.00022089604],"domain_scores_gemma":[0.9905934,0.005397995,0.0006125016,0.0015138785,0.001565166,0.0003171171],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054090726,0.0036940724,0.001785994,0.0027669722,0.0006798228,0.0016823554,0.0051355194,0.002027119,0.0038030974],"category_scores_gemma":[0.020582594,0.0011162322,0.0016467323,0.0010531729,0.0009281341,0.003077069,0.0031397452,0.0031946886,0.0019736588],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009997992,0.00056040747,0.024298042,0.0005042787,0.00071890384,0.00067388854,0.00025380278,0.35096568,0.02146559,0.0034125322,0.016282888,0.5798642],"study_design_scores_gemma":[0.00008312088,0.00024267017,0.001281036,0.00004265992,0.00008688369,0.00014830157,0.00006121718,0.9818594,0.008856926,0.005247846,0.002056417,0.00003348927],"about_ca_topic_score_codex":0.0060395906,"about_ca_topic_score_gemma":0.014204347,"teacher_disagreement_score":0.0060395906,"about_ca_system_score_codex":0.0014696501,"about_ca_system_score_gemma":0.0023130598,"threshold_uncertainty_score":0.028606296},"labels":[],"label_agreement":null},{"id":"W4414682256","doi":"10.48550/arxiv.2506.19045","title":"Efficient Black-Box Fault Localization for System-Level Test Code Using Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code (set theory); TRACE (psycholinguistics); Pruning; Test suite; Code coverage; Inference; Source code; Test case; Debugging; Fault (geology)","score_opus":0.10090066710501147,"score_gpt":0.2378397593333464,"score_spread":0.13693909222833492,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414682256","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1215538,0.001353726,0.776191,0.00063101196,0.000091834496,0.0002487736,0.0024139567,0.09643012,0.0010857129],"genre_scores_gemma":[0.56622803,0.00035803602,0.4158059,0.00045712,0.000048166403,0.0003684443,0.011385298,0.0031749548,0.0021740135],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99819344,0.00043244858,0.00012040026,0.00062022125,0.00048431297,0.00014917612],"domain_scores_gemma":[0.99189633,0.0049573174,0.0009780211,0.0010745713,0.0008785509,0.00021526193],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013686867,0.0027102295,0.0012275378,0.0026440835,0.00055969413,0.0013218796,0.0024781877,0.0014798463,0.0018580704],"category_scores_gemma":[0.011370046,0.00081583613,0.0020050548,0.0010194669,0.00085913186,0.0021732065,0.0016494556,0.0022355001,0.002500746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078537164,0.0006586273,0.031374946,0.0010314295,0.00025683173,0.0010377411,0.0007374894,0.2972985,0.041563913,0.0038063757,0.018837754,0.60261095],"study_design_scores_gemma":[0.000029725226,0.00009513796,0.0010221659,0.00003293912,0.000026447973,0.00015710959,0.00008038006,0.985104,0.008844126,0.0030167555,0.0015734133,0.000017794775],"about_ca_topic_score_codex":0.009287605,"about_ca_topic_score_gemma":0.018016398,"teacher_disagreement_score":0.009287605,"about_ca_system_score_codex":0.0012512411,"about_ca_system_score_gemma":0.0022347185,"threshold_uncertainty_score":0.018467069},"labels":[],"label_agreement":null},{"id":"W4414988640","doi":"10.1145/3763160","title":"Products of Recursive Programs for Hypersafety Verification","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Key (lock); Product (mathematics); Class (philosophy); Property (philosophy); Set (abstract data type); Simple (philosophy); Parametric statistics","score_opus":0.021377923818186817,"score_gpt":0.2885382608968596,"score_spread":0.26716033707867276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414988640","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018584179,0.00006711194,0.97809863,0.00013633232,0.000013178338,0.000045797893,0.000026580967,0.000771762,0.002256345],"genre_scores_gemma":[0.4438442,0.00022548741,0.55202144,0.00016391283,0.00004865962,0.00026778417,0.00019376921,0.0005312566,0.0027034467],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99721706,0.0009737871,0.00015096793,0.00056560466,0.0008569032,0.00023569402],"domain_scores_gemma":[0.9900739,0.006138937,0.00057079067,0.0023575574,0.000698777,0.00015998672],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034355593,0.00062500255,0.00050163764,0.000653406,0.000875892,0.001935098,0.0018265862,0.000967332,0.003689437],"category_scores_gemma":[0.011246799,0.0006347776,0.0011753506,0.00063388195,0.0055625085,0.007438308,0.0031117792,0.00276215,0.00064449373],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009456359,0.000049524508,0.0010061477,0.000119754775,0.000020419006,0.00015516512,0.0005913529,0.02722711,0.00612,0.91081196,0.00069266924,0.053111356],"study_design_scores_gemma":[0.000030043093,0.00011032348,0.00021944108,0.000049786384,0.000044738426,0.00023691759,0.000106134714,0.218098,0.022977836,0.74510926,0.012980479,0.00003706209],"about_ca_topic_score_codex":0.0011670447,"about_ca_topic_score_gemma":0.0010119452,"teacher_disagreement_score":0.003689437,"about_ca_system_score_codex":0.0009020456,"about_ca_system_score_gemma":0.001511751,"threshold_uncertainty_score":0.018169165},"labels":[],"label_agreement":null},{"id":"W4415007091","doi":"10.1145/3763053","title":"Boosting Program Reduction with the Missing Piece of Syntax-Guided Transformations","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Debugging; Reduction (mathematics); Boosting (machine learning); Leverage (statistics); Program analysis; Program transformation; Key (lock); Abstract syntax tree; Minification","score_opus":0.01579920877483586,"score_gpt":0.2941947941011593,"score_spread":0.2783955853263234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415007091","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057258133,0.0008424531,0.8628424,0.0007644774,0.00014695074,0.00034398155,0.00040750497,0.068768926,0.008625103],"genre_scores_gemma":[0.2826532,0.00050300674,0.6992985,0.0009260749,0.00008844716,0.00040486007,0.0019289427,0.008626326,0.00557065],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9959422,0.00085070083,0.00029008408,0.00074426836,0.0018419492,0.00033082676],"domain_scores_gemma":[0.9936783,0.001963788,0.000424145,0.0026259231,0.0011863052,0.00012145088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016799681,0.0017550805,0.0009069817,0.00196167,0.000577136,0.0015369955,0.0034963524,0.00090035703,0.003866569],"category_scores_gemma":[0.009049305,0.000738157,0.001872412,0.0015286424,0.0021810876,0.00385475,0.0028214937,0.0027381682,0.0020488258],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007299621,0.0005874992,0.007823786,0.0012227586,0.00021042768,0.00048689698,0.0008755303,0.043559965,0.109104864,0.059512176,0.02920133,0.7466848],"study_design_scores_gemma":[0.00042296157,0.00092301314,0.002794931,0.00016247664,0.00036921527,0.0014237267,0.0004066282,0.544855,0.2818308,0.06728446,0.099280186,0.00024663037],"about_ca_topic_score_codex":0.002350986,"about_ca_topic_score_gemma":0.0036426783,"teacher_disagreement_score":0.003866569,"about_ca_system_score_codex":0.00089320214,"about_ca_system_score_gemma":0.00281316,"threshold_uncertainty_score":0.012934983},"labels":[],"label_agreement":null},{"id":"W4416240214","doi":"10.1007/s10664-025-10700-7","title":"Fuzzing-based mutation testing of C/C++ software in cyber-physical systems","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Kangwon National University; European Space Agency","keywords":"Fuzz testing; Symbolic execution; Code coverage; Process (computing); Software testing; Mutation testing; Mutation; Software; Software bug; Test case","score_opus":0.02328434506771092,"score_gpt":0.2833576778545125,"score_spread":0.26007333278680156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416240214","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8114405,0.00019225274,0.1853522,0.00019819401,0.000023754956,0.000055828335,0.00010820975,0.0012850832,0.0013438881],"genre_scores_gemma":[0.974596,0.000022577233,0.025157347,0.000016967491,0.0000030035155,0.00001155737,0.000040922845,0.000027770018,0.00012383395],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.99802256,0.00075496105,0.00009510081,0.00026955057,0.0007439937,0.00011397034],"domain_scores_gemma":[0.9760105,0.018568156,0.0014128898,0.002074826,0.0017047501,0.0002289185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024873682,0.00048239154,0.0003978202,0.001284132,0.00034127338,0.0006277894,0.0012164233,0.0007849872,0.00087167334],"category_scores_gemma":[0.026751306,0.0002443078,0.00035973525,0.0006426659,0.0008606181,0.0012676222,0.0005497062,0.0007715505,0.00008394055],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009383833,0.00060443045,0.05067787,0.00026718006,0.00017676382,0.00038829062,0.000471059,0.6771694,0.047043473,0.022047488,0.0009372313,0.19927847],"study_design_scores_gemma":[0.000017798353,0.00008425346,0.0046571046,0.000011505515,0.000019797506,0.00008405843,0.000019489848,0.98328847,0.0083233975,0.0034012597,0.00008390314,0.0000089438745],"about_ca_topic_score_codex":0.0049316445,"about_ca_topic_score_gemma":0.005208926,"teacher_disagreement_score":0.0049316445,"about_ca_system_score_codex":0.00082732196,"about_ca_system_score_gemma":0.00095370214,"threshold_uncertainty_score":0.013154566},"labels":[],"label_agreement":null},{"id":"W4416248181","doi":"10.1007/s10664-025-10761-8","title":"SBEST: Spectrum-based fault localization without fault-triggering tests","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Concordia University","funders":"","keywords":"Stack (abstract data type); Software bug; Context (archaeology); TRACE (psycholinguistics); Call stack; Fault (geology); Software regression; Software; Debugging","score_opus":0.01565781770040737,"score_gpt":0.28534965561632986,"score_spread":0.2696918379159225,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416248181","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021138906,0.0001267535,0.92196953,0.000091386675,0.000075850694,0.000093112,0.0004106087,0.05447465,0.0016191663],"genre_scores_gemma":[0.4611116,0.000089193774,0.5311285,0.00017141504,0.00005888853,0.00017767829,0.0011953518,0.003153414,0.0029139263],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997171,0.00066656666,0.00016352188,0.00048386765,0.0012744422,0.00024066046],"domain_scores_gemma":[0.9920861,0.0029980708,0.0007610139,0.0026441007,0.00118563,0.0003250115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016819936,0.001837779,0.001408279,0.0032084328,0.00059793715,0.0011003227,0.003098178,0.0013632729,0.006734475],"category_scores_gemma":[0.009959416,0.0007178409,0.0008673638,0.001345972,0.0011821279,0.0027039896,0.0024659014,0.0014035439,0.0025714003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030242505,0.000813999,0.009108804,0.00070888846,0.000300243,0.00069709064,0.00032173243,0.10135403,0.06372995,0.021027833,0.021091642,0.77782154],"study_design_scores_gemma":[0.0002564294,0.00048989785,0.0018014507,0.00006207136,0.000102730584,0.0005030567,0.000068212306,0.92137897,0.044135667,0.026951576,0.004184089,0.00006586153],"about_ca_topic_score_codex":0.0021200327,"about_ca_topic_score_gemma":0.002619432,"teacher_disagreement_score":0.006734475,"about_ca_system_score_codex":0.0005111044,"about_ca_system_score_gemma":0.0012629853,"threshold_uncertainty_score":0.022529006},"labels":[],"label_agreement":null},{"id":"W4416298547","doi":"10.4204/eptcs.436.5","title":"Mutation Testing for Industrial Robotic Systems","year":2025,"lang":"en","type":"article","venue":"Electronic Proceedings in Theoretical Computer Science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Chicoutimi","funders":"","keywords":"Mutation; Software; Reliability (semiconductor); Process (computing); Robot; Industrial robot; Quality (philosophy); Mutation testing","score_opus":0.020033874459946397,"score_gpt":0.2757728134674147,"score_spread":0.2557389390074683,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416298547","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20352332,0.00081626023,0.78978187,0.00047949876,0.00005903053,0.00013709652,0.000099924924,0.0026934484,0.002409615],"genre_scores_gemma":[0.8201856,0.00033012615,0.17796132,0.00017093935,0.00002610712,0.0001186899,0.00014665935,0.00020078062,0.0008597942],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99638474,0.0015323408,0.00017713114,0.00038323464,0.0013454937,0.00017700279],"domain_scores_gemma":[0.9909051,0.006850026,0.00076022826,0.0006310994,0.00070417655,0.00014938977],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020178824,0.00075773796,0.0005060356,0.0012439644,0.0004033975,0.0006377906,0.0011905981,0.0009107127,0.0007573806],"category_scores_gemma":[0.012368875,0.00021080926,0.0006078928,0.0007357225,0.0018851991,0.0009522988,0.00083662546,0.00090071955,0.00011563826],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003883522,0.0003424264,0.011500484,0.0005548734,0.00010530093,0.0017015208,0.0005171717,0.55903476,0.0962648,0.06353067,0.0016440778,0.26441562],"study_design_scores_gemma":[0.00006747619,0.00033714346,0.0015885863,0.000069981834,0.000040992236,0.00069878495,0.00006682478,0.9003398,0.04369273,0.05011024,0.002950183,0.000037229864],"about_ca_topic_score_codex":0.00198539,"about_ca_topic_score_gemma":0.0013130934,"teacher_disagreement_score":0.0020178824,"about_ca_system_score_codex":0.00088035414,"about_ca_system_score_gemma":0.00086839334,"threshold_uncertainty_score":0.010671735},"labels":[],"label_agreement":null},{"id":"W4416368800","doi":"10.48550/arxiv.2510.02534","title":"ZeroFalse: Improving Precision in Static Analysis with LLMs","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Static analysis; False positive paradox; Benchmark (surveying); Reliability (semiconductor); Precision and recall; Static program analysis; Software; Java","score_opus":0.03196078289550708,"score_gpt":0.2928788205595702,"score_spread":0.26091803766406313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416368800","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05452817,0.003431528,0.64806324,0.0017772025,0.00033244272,0.00026047917,0.0029466744,0.28382567,0.0048346105],"genre_scores_gemma":[0.3609785,0.00081786414,0.61212814,0.0012345137,0.00020799664,0.00024777395,0.008159427,0.012693002,0.0035327086],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9754009,0.009066128,0.0019861688,0.005185272,0.007438681,0.0009228755],"domain_scores_gemma":[0.9208629,0.048261747,0.0040638526,0.019296113,0.006831915,0.0006835405],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019633034,0.0042678127,0.0021266292,0.008476991,0.0012264585,0.0061691967,0.0053284965,0.002745902,0.0035609724],"category_scores_gemma":[0.09834177,0.0018390311,0.0029098352,0.0037337667,0.002269339,0.011792893,0.0069591515,0.0036699488,0.0041547087],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015210657,0.00064648746,0.04662479,0.002215484,0.0007684814,0.00051715,0.0020206172,0.05840017,0.030478103,0.020997219,0.07420255,0.76160794],"study_design_scores_gemma":[0.00026035064,0.00047746982,0.005157977,0.000513325,0.00041366022,0.0007532289,0.0004889533,0.82771045,0.05576739,0.059986163,0.048182555,0.0002884951],"about_ca_topic_score_codex":0.0096849,"about_ca_topic_score_gemma":0.017189005,"teacher_disagreement_score":0.019633034,"about_ca_system_score_codex":0.0022391756,"about_ca_system_score_gemma":0.006000287,"threshold_uncertainty_score":0.103830695},"labels":[],"label_agreement":null},{"id":"W4416799210","doi":"10.1109/snpd65828.2025.11252591","title":"Making the Case for LLM-Generated Automated Program Repair Benchmarks","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Algoma University","funders":"","keywords":"Overfitting; Benchmark (surveying); Benchmarking; Focus (optics); Quality (philosophy); Quality assurance","score_opus":0.04917975613988371,"score_gpt":0.3669347614918263,"score_spread":0.31775500535194257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416799210","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24314317,0.0048488327,0.65258825,0.02113072,0.0020889323,0.0013259118,0.0070533543,0.041543707,0.026277216],"genre_scores_gemma":[0.5088752,0.00061735325,0.4687972,0.004097927,0.00019210514,0.0014237844,0.010528454,0.0034220999,0.0020458237],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.96848047,0.02036177,0.0017852619,0.0023452695,0.006338808,0.0006885293],"domain_scores_gemma":[0.8848275,0.064278,0.0054100435,0.027534999,0.016346741,0.0016027602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026681,0.0011531134,0.00069792103,0.0023681642,0.00077397836,0.0032820173,0.0036683448,0.0018323794,0.0017918983],"category_scores_gemma":[0.18955101,0.00057657156,0.0006616949,0.0019566212,0.0019457883,0.004565858,0.0033999488,0.0035886918,0.0010651702],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015753583,0.0015301015,0.046665456,0.002297729,0.00039182013,0.0005645339,0.001643969,0.21038695,0.02232483,0.086331345,0.12822123,0.49806678],"study_design_scores_gemma":[0.00059573865,0.0012583133,0.010824053,0.0011941459,0.00013594047,0.00035714076,0.0008574432,0.7490356,0.034014266,0.11783318,0.083743915,0.00015023867],"about_ca_topic_score_codex":0.0030081703,"about_ca_topic_score_gemma":0.0064831413,"teacher_disagreement_score":0.026681,"about_ca_system_score_codex":0.001798524,"about_ca_system_score_gemma":0.0032326074,"threshold_uncertainty_score":0.14110434},"labels":[],"label_agreement":null},{"id":"W4416961966","doi":"10.1109/pst65910.2025.11268838","title":"RefPentester: A Knowledge-Informed Self-Reflective Penetration Testing Framework Based on Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Concordia University","funders":"Concordia University","keywords":"Process (computing); Hacker; Baseline (sea); Fuzz testing; Soundness","score_opus":0.032705316852876025,"score_gpt":0.3436117691507119,"score_spread":0.31090645229783587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416961966","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00677505,0.00021224044,0.9566146,0.00038296485,0.000030963743,0.00024624175,0.00025066323,0.03353546,0.001951847],"genre_scores_gemma":[0.28592268,0.00032637038,0.7056496,0.000588985,0.000034019293,0.0006003071,0.0013571138,0.0024206035,0.0031003573],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99735487,0.0011893553,0.00015018601,0.0004748134,0.0006418177,0.00018897447],"domain_scores_gemma":[0.9933468,0.0042219185,0.0005170785,0.0011819395,0.000524405,0.00020785657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003127968,0.002006563,0.00076331774,0.0012764643,0.0004974408,0.0018083035,0.003919564,0.001663826,0.005398311],"category_scores_gemma":[0.014998508,0.0010558773,0.0019607476,0.00047444593,0.0017560307,0.0050669815,0.0038764311,0.002889922,0.0015899923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004333526,0.00067050004,0.0071357545,0.0008856354,0.00022580558,0.0010480775,0.0013527832,0.4794097,0.012993148,0.05290028,0.018918522,0.42402637],"study_design_scores_gemma":[0.000033978595,0.00008535808,0.00022933353,0.00004897875,0.00002409583,0.00017800962,0.000054543325,0.9668605,0.0033582123,0.02266419,0.006429221,0.00003366501],"about_ca_topic_score_codex":0.0066413833,"about_ca_topic_score_gemma":0.0127584785,"teacher_disagreement_score":0.0066413833,"about_ca_system_score_codex":0.0011129673,"about_ca_system_score_gemma":0.0026764544,"threshold_uncertainty_score":0.018059134},"labels":[],"label_agreement":null},{"id":"W4417036956","doi":"10.5753/sbqs.2025.15010","title":"Automated Test Case Generation in a Real-World System Using a Customized AI Agent: An Experience Report","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Optech (Canada)","funders":"","keywords":"Context (archaeology); Component (thermodynamics); Process (computing); Quality (philosophy); Test (biology); Test case; Software; Keyword-driven testing; Test Management Approach","score_opus":0.06728239215720491,"score_gpt":0.3727870787301468,"score_spread":0.30550468657294194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417036956","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.83104414,0.0002037706,0.15707955,0.00032278744,0.000033934477,0.0008158205,0.00019585405,0.002410394,0.007893619],"genre_scores_gemma":[0.75053555,0.00022764286,0.24459669,0.00009590379,0.000014828049,0.00025561752,0.0005915941,0.00030749978,0.0033747018],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99666446,0.00204055,0.0001967673,0.00031248538,0.00060931407,0.0001764211],"domain_scores_gemma":[0.9864325,0.0097342385,0.00038751654,0.0016471612,0.0013296297,0.00046906143],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005719208,0.0008290799,0.0004297237,0.00063399854,0.0004254932,0.0010425354,0.0018487409,0.0010247857,0.0018977581],"category_scores_gemma":[0.012997042,0.00042089488,0.00042377427,0.00048268083,0.0009135677,0.0010076944,0.0009938788,0.0009032179,0.0007174456],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001827521,0.009677382,0.027129982,0.0010586861,0.0002446754,0.005066262,0.019401707,0.11477094,0.16126563,0.005640739,0.007991759,0.64592475],"study_design_scores_gemma":[0.0011192613,0.013685136,0.033150773,0.00029076467,0.00032155018,0.0058126957,0.0050015682,0.5331912,0.33716288,0.0027348788,0.06718323,0.00034603724],"about_ca_topic_score_codex":0.00208374,"about_ca_topic_score_gemma":0.0026750013,"teacher_disagreement_score":0.005719208,"about_ca_system_score_codex":0.00065920106,"about_ca_system_score_gemma":0.00071118126,"threshold_uncertainty_score":0.030246437},"labels":[],"label_agreement":null},{"id":"W4417132523","doi":"10.1109/saner-c66551.2025.00015","title":"Assessing Data Augmentation-Induced Bias in Training and Testing of Machine Learning Models","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Debugging; Training set; Software; Test data; Test (biology); Training (meteorology); Non-regression testing; Keyword-driven testing; Software testing","score_opus":0.4312538849540782,"score_gpt":0.4003276714823415,"score_spread":0.030926213471736685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417132523","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.69085735,0.0027503602,0.29588932,0.0017631456,0.0004235237,0.0004992119,0.0016887614,0.0026028047,0.003525513],"genre_scores_gemma":[0.9005082,0.00023151928,0.0944933,0.00061168824,0.00009706894,0.00034318984,0.0027647822,0.00032296075,0.0006273964],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9754882,0.015567099,0.0015659723,0.0026310587,0.0041335346,0.00061417173],"domain_scores_gemma":[0.7182247,0.22836292,0.009900182,0.028344866,0.01388572,0.0012816174],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.040821016,0.0015852298,0.0012780403,0.0014271514,0.00099545,0.0020402963,0.0018548543,0.002555319,0.0010692618],"category_scores_gemma":[0.22313894,0.0007571227,0.0011522685,0.0012712472,0.0030099512,0.0029867892,0.003613195,0.003451975,0.0005093538],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037215883,0.00090212486,0.15785562,0.0010405405,0.0011028211,0.00047150927,0.0012444943,0.61772764,0.014362178,0.012105031,0.008026217,0.18144023],"study_design_scores_gemma":[0.00026796927,0.0011633667,0.021190004,0.0003524809,0.00021975278,0.00039673335,0.00030654023,0.921308,0.027462998,0.02291215,0.004329377,0.00009070882],"about_ca_topic_score_codex":0.0028553011,"about_ca_topic_score_gemma":0.0035155087,"teacher_disagreement_score":0.959179,"about_ca_system_score_codex":0.001216545,"about_ca_system_score_gemma":0.0019059505,"threshold_uncertainty_score":0.21588475},"labels":[],"label_agreement":null},{"id":"W4417436230","doi":"10.1145/3785363","title":"Galápagos: Automated N-Version Programming with LLMs","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Correctness; Redundancy (engineering); Software; Symbolic execution; Program analysis; Automatic programming; Software quality; Programming paradigm; Software testing","score_opus":0.04061881690512797,"score_gpt":0.3080365130385907,"score_spread":0.2674176961334627,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417436230","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09482519,0.000117183285,0.8746911,0.00013430468,0.000029432622,0.00014223276,0.00018754628,0.027203469,0.0026695246],"genre_scores_gemma":[0.5407792,0.00009174553,0.4516693,0.00017210525,0.000013607708,0.00023227352,0.0005250438,0.004131954,0.0023848142],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983918,0.00049220084,0.000105852894,0.0003137095,0.0005590653,0.0001374139],"domain_scores_gemma":[0.99545527,0.0018339498,0.00038780548,0.0019222928,0.00030963458,0.000091042995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017044256,0.00070435327,0.0003685252,0.0006781533,0.00040687277,0.0010464037,0.002145301,0.0007361535,0.0023771136],"category_scores_gemma":[0.0060685817,0.00071616174,0.0011118911,0.00033015021,0.0017450234,0.0019254384,0.0024363734,0.0011708032,0.0006497688],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013746363,0.00043019143,0.024629148,0.00094312715,0.00023312731,0.0015778892,0.0022374557,0.28150395,0.18662642,0.10457791,0.00835447,0.38751167],"study_design_scores_gemma":[0.00014773395,0.00040792127,0.0017081473,0.00010457739,0.00007248149,0.0007491267,0.0001248392,0.77795863,0.14649852,0.053163916,0.018972997,0.00009109605],"about_ca_topic_score_codex":0.0009956084,"about_ca_topic_score_gemma":0.0013415407,"teacher_disagreement_score":0.0023771136,"about_ca_system_score_codex":0.0006688661,"about_ca_system_score_gemma":0.0010594856,"threshold_uncertainty_score":0.009013951},"labels":[],"label_agreement":null},{"id":"W48642625","doi":"","title":"Validation Against Actual Behavior: Still a Challenge for Testing Tools.","year":2010,"lang":"en","type":"article","venue":"Software Engineering Research and Practice","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University; Carleton University","funders":"","keywords":"Computer science; Traceability; Software engineering; Acceptance testing; Model-based testing; Quality (philosophy); Code (set theory); Test (biology); Test case; Programming language; Machine learning","score_opus":0.13507361304108773,"score_gpt":0.3782731407054294,"score_spread":0.24319952766434166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W48642625","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038683396,0.0020237274,0.94240385,0.0053992216,0.0002620788,0.00037124893,0.00017562816,0.00441554,0.0062654084],"genre_scores_gemma":[0.36961338,0.0021894965,0.6204802,0.0024313196,0.00021797542,0.00077643746,0.00076618843,0.001574009,0.0019509958],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.91697234,0.034395747,0.0054701073,0.004881979,0.03698991,0.0012899232],"domain_scores_gemma":[0.63561374,0.23367815,0.018872002,0.08119463,0.028753923,0.0018876294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.055964343,0.0018379373,0.0017915205,0.002305927,0.0013071059,0.0067548295,0.0058455,0.0043440172,0.0016632187],"category_scores_gemma":[0.2392383,0.000960637,0.0012309966,0.001631183,0.006379229,0.012524098,0.0045124968,0.005194858,0.001224063],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026351883,0.00081254676,0.029661004,0.0038232193,0.00053294847,0.0010425921,0.008017714,0.03456453,0.03855792,0.21655717,0.010140682,0.6560261],"study_design_scores_gemma":[0.00033896416,0.003069059,0.016488561,0.008726912,0.00071178493,0.007146618,0.0060652676,0.25005573,0.095394015,0.47768968,0.13357861,0.0007347598],"about_ca_topic_score_codex":0.0013873118,"about_ca_topic_score_gemma":0.0009476159,"teacher_disagreement_score":0.055964343,"about_ca_system_score_codex":0.0017360077,"about_ca_system_score_gemma":0.0052247182,"threshold_uncertainty_score":0.2959712},"labels":[],"label_agreement":null},{"id":"W49013149","doi":"","title":"Software Testing and Quality Assurance","year":2007,"lang":"en","type":"book","venue":"John Wiley & Sons eBooks","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Quality assurance; Software quality assurance; Software quality; Software quality analyst; Computer science; Software engineering; Software; Reliability engineering; Engineering; Operations management; Operating system; Software development; External quality assessment","score_opus":0.05730441950692096,"score_gpt":0.29686374559851847,"score_spread":0.23955932609159752,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W49013149","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021330605,0.097194485,0.25131953,0.0026386657,0.002146888,0.00024203946,0.00064975146,0.0052427165,0.63843274],"genre_scores_gemma":[0.018189916,0.042831317,0.090427935,0.0012857274,0.0007637971,0.00035025485,0.0011252032,0.0020562427,0.84296954],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99858534,0.00014551292,0.000058274956,0.00013343278,0.0010259263,0.000051569357],"domain_scores_gemma":[0.9982393,0.00074556854,0.000092817274,0.0003472371,0.00050862966,0.00006644698],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008596321,0.0020629417,0.001913682,0.0030655642,0.00062023225,0.0024651506,0.0013617014,0.0016179396,0.04723322],"category_scores_gemma":[0.0031275158,0.0010365078,0.00048351847,0.0041601337,0.0012364078,0.003163572,0.001644686,0.0025629716,0.03669216],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015987122,0.00005186867,0.00010424913,0.0003981588,0.0000129546215,0.00005642621,0.00016218956,0.0014907074,0.0012323662,0.030333845,0.1458801,0.8202611],"study_design_scores_gemma":[0.000017920953,0.00008380823,0.000769892,0.00080599776,0.00003628476,0.0006738224,0.00008261031,0.0050454456,0.0018562521,0.069692776,0.9209013,0.000033766904],"about_ca_topic_score_codex":0.0019459631,"about_ca_topic_score_gemma":0.004048902,"teacher_disagreement_score":0.04723322,"about_ca_system_score_codex":0.00089500233,"about_ca_system_score_gemma":0.0011738412,"threshold_uncertainty_score":0.1580109},"labels":[],"label_agreement":null},{"id":"W49830884","doi":"10.1007/978-3-642-36054-1_7","title":"People-Centered Software Development: An Overview of Agile Methodologies","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Agile software development; Computer science; Agile Unified Process; Agile usability engineering; Software engineering; Focus (optics); Test-driven development; Software development; Quality assurance; Lean software development; Systems engineering; Software; Software development process; Engineering; Programming language; Operations management","score_opus":0.14082622124068347,"score_gpt":0.33964033243070785,"score_spread":0.19881411119002437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W49830884","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004551576,0.35762572,0.56622654,0.004102828,0.00077191304,0.00023424109,0.00011063446,0.0010289343,0.065347604],"genre_scores_gemma":[0.058833763,0.38393933,0.52118415,0.0019463119,0.0007001613,0.00042926476,0.00031443383,0.00034298404,0.032309603],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981188,0.000551728,0.00014090959,0.00017015137,0.0008855255,0.0001329065],"domain_scores_gemma":[0.99861,0.0008038658,0.00010012628,0.000106923086,0.00027795168,0.00010107059],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019150375,0.0007503844,0.00068444625,0.0024232191,0.00054288254,0.002619847,0.0011598129,0.0014164338,0.0025572293],"category_scores_gemma":[0.0016720863,0.0007288491,0.0006032104,0.0048773084,0.0010798497,0.002615936,0.0015992237,0.0022017676,0.0016085268],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020016185,0.00013294579,0.00046337675,0.0021219717,0.000036590714,0.00012638028,0.0010770878,0.0020166808,0.0023881292,0.07060091,0.0156433,0.9053726],"study_design_scores_gemma":[0.000025540126,0.00017359486,0.0012017647,0.0030824144,0.00003487966,0.0012486952,0.0005625294,0.005290607,0.0033776914,0.103414126,0.88153505,0.00005314711],"about_ca_topic_score_codex":0.0009468983,"about_ca_topic_score_gemma":0.0014160568,"teacher_disagreement_score":0.002619847,"about_ca_system_score_codex":0.00089008466,"about_ca_system_score_gemma":0.001742486,"threshold_uncertainty_score":0.010127783},"labels":[],"label_agreement":null},{"id":"W58067616","doi":"10.1007/978-3-642-35267-6_13","title":"Regression Testing of Object-Oriented Software: A Technique Based on Use Cases and Associated Tool","year":2012,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Regression testing; Computer science; Software; Information retrieval; Data mining; Software engineering; Programming language; Software development; Software construction","score_opus":0.07075414036268389,"score_gpt":0.30479052641519233,"score_spread":0.23403638605250843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W58067616","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010741645,0.0006406887,0.9745377,0.00016336424,0.00004600085,0.00010021163,0.00006239812,0.0039629894,0.009745008],"genre_scores_gemma":[0.16178547,0.0013145038,0.82488525,0.0001085894,0.00006844703,0.00021142709,0.0002730096,0.0013883545,0.0099649755],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9981254,0.0005569235,0.00011866353,0.0001776087,0.0009508911,0.00007052554],"domain_scores_gemma":[0.9938811,0.0044989577,0.0003626431,0.000754406,0.00043921138,0.00006367332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013120195,0.0012862441,0.00059999945,0.002464376,0.00044314007,0.0011504266,0.0022211475,0.00096943096,0.003974374],"category_scores_gemma":[0.007140226,0.0008430628,0.0009055561,0.002113336,0.0009964814,0.0020538042,0.000955274,0.0017972336,0.0010444332],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011998143,0.00027762912,0.002261526,0.0006145718,0.000068587935,0.0012506632,0.0007753708,0.012607814,0.041822318,0.056306664,0.007074056,0.8768208],"study_design_scores_gemma":[0.00009930237,0.0006512744,0.008287124,0.0008434281,0.0004572132,0.011940709,0.0003977675,0.5824846,0.17829128,0.117783315,0.09857399,0.00018993508],"about_ca_topic_score_codex":0.0006661269,"about_ca_topic_score_gemma":0.0011910525,"teacher_disagreement_score":0.003974374,"about_ca_system_score_codex":0.0002556396,"about_ca_system_score_gemma":0.000427425,"threshold_uncertainty_score":0.013295591},"labels":[],"label_agreement":null},{"id":"W6191748","doi":"10.1111/j.1440-1754.1983.tb02042.x","title":"Generative hierarchical contracts for conformance testing of sequential containers","year":2007,"lang":"en","type":"article","venue":"Australian paediatric journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Domain (mathematical analysis); Component (thermodynamics); Generative grammar; Domain analysis; Conformance testing; Domain model; Fuzz testing; Unit testing; Generative model; Software testing; Software engineering; Model-based testing; Software; Programming language; Artificial intelligence; Software development; Domain knowledge; Test case; Machine learning; Mathematics; Software construction","score_opus":0.052735595578852996,"score_gpt":0.31585327629823523,"score_spread":0.26311768071938224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6191748","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43574178,0.00013597937,0.5498142,0.00036728717,0.00003320672,0.00034952495,0.00061669084,0.0016314046,0.011309863],"genre_scores_gemma":[0.8578012,0.000034484678,0.13901414,0.00003772282,0.000012916297,0.00015822187,0.00057740964,0.00014911416,0.002214698],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9913453,0.0036233792,0.0006765807,0.00094246917,0.0025861599,0.00082617137],"domain_scores_gemma":[0.947483,0.03692017,0.0037165582,0.008012362,0.002871408,0.0009965494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008246235,0.000332714,0.00046053037,0.0017448938,0.0008548085,0.0015401066,0.0018678085,0.0008701535,0.005516968],"category_scores_gemma":[0.05185977,0.00053680467,0.0012072807,0.0017300962,0.003940012,0.0028327478,0.0025844024,0.0012075573,0.0004629514],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011078372,0.00032658075,0.07077612,0.00022865534,0.00008361508,0.0013976975,0.0044483803,0.1007871,0.008834382,0.53406644,0.0037910508,0.2741521],"study_design_scores_gemma":[0.00008556715,0.00022021578,0.01653106,0.00005624844,0.00003310171,0.0007537666,0.00075108,0.58455175,0.0054463763,0.38703468,0.004453301,0.00008285168],"about_ca_topic_score_codex":0.009811869,"about_ca_topic_score_gemma":0.011126945,"teacher_disagreement_score":0.009811869,"about_ca_system_score_codex":0.0020560883,"about_ca_system_score_gemma":0.0021291138,"threshold_uncertainty_score":0.04361075},"labels":[],"label_agreement":null},{"id":"W658116578","doi":"","title":"SToP : Scalable Termination analysis of (C) Programs (tool presentation)","year":2012,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Emera (Canada)","funders":"","keywords":"Program slicing; Computer science; Slicing; Scalability; Modular design; Programming language; Program analysis; Code (set theory); Static analysis; Operating system; World Wide Web","score_opus":0.022731433104984475,"score_gpt":0.262514607741428,"score_spread":0.2397831746364435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W658116578","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026957018,0.00020646672,0.90950567,0.00042623153,0.00015012152,0.00016723963,0.0007493238,0.057716444,0.0041214717],"genre_scores_gemma":[0.4879116,0.00026314365,0.48233324,0.0005669221,0.000334206,0.00049207674,0.0037839618,0.015298805,0.00901598],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967188,0.0006451961,0.00017426819,0.0006082511,0.0013622612,0.0004911095],"domain_scores_gemma":[0.9912698,0.004997754,0.00051958533,0.0016415104,0.0011833593,0.00038798558],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026075607,0.0023066886,0.0016268573,0.0017379622,0.0011965419,0.0023572138,0.0034420385,0.0016358505,0.01588163],"category_scores_gemma":[0.010264358,0.00086373586,0.00217756,0.0013019274,0.0018557778,0.0032458464,0.0032057485,0.003269969,0.003371302],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037923718,0.00057137886,0.009679705,0.0017024968,0.00039395085,0.0014457062,0.0009809347,0.17085758,0.09044331,0.08440922,0.12663369,0.50908965],"study_design_scores_gemma":[0.0004992569,0.0003631278,0.0017192682,0.00013635903,0.00013943434,0.0002913234,0.00013417534,0.8552733,0.037716273,0.09275573,0.010890012,0.000081677535],"about_ca_topic_score_codex":0.0037920699,"about_ca_topic_score_gemma":0.004233304,"teacher_disagreement_score":0.01588163,"about_ca_system_score_codex":0.0010544674,"about_ca_system_score_gemma":0.0020489283,"threshold_uncertainty_score":0.053129315},"labels":[],"label_agreement":null},{"id":"W6889756539","doi":"10.26262/heal.auth.ir.286909","title":"Φυσικοθεραπευτικές τεχνικές για την αποκατάσταση του επώδυνου ημιπληγικού ώμου μετά από εγκεφαλικό επεισόδιο: Συστηματική ανασκόπηση και μετα-ανάλυση","year":2015,"lang":"el","type":"article","venue":"Aristotle University of Thessaloniki","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Stroke (engine); MEDLINE; Cochrane collaboration; Systematic review; Meta-analysis; Clinical trial","score_opus":0.06140312759617738,"score_gpt":0.21973553710706073,"score_spread":0.15833240951088334,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6889756539","genre_codex":"other","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05731322,0.0064872205,0.091612376,0.026888765,0.0029760073,0.00026206576,0.0012627114,0.0013887984,0.81180876],"genre_scores_gemma":[0.4965749,0.0077587715,0.047150645,0.0057657664,0.0010570047,0.00049324357,0.0012378781,0.0015849402,0.43837687],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"meta_analysis","domain_scores_codex":[0.99706596,0.0004750728,0.000116837684,0.0007109312,0.0012505691,0.00038067705],"domain_scores_gemma":[0.9958917,0.001146783,0.00035810797,0.0006169573,0.0014121796,0.0005743709],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019178861,0.00088793266,0.0006307013,0.0013712759,0.0031824093,0.007765476,0.001416518,0.0030629057,0.085562505],"category_scores_gemma":[0.008225355,0.0006708519,0.0006606263,0.0012576147,0.0043151937,0.007988554,0.0038941072,0.003611218,0.030688673],"study_design_candidate":"meta_analysis","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005942717,0.000324466,0.004547809,0.0015929675,0.00009318794,0.0016778886,0.02201054,0.001676386,0.04633634,0.62995285,0.08860074,0.20259261],"study_design_scores_gemma":[0.000058922305,0.00011509811,0.0057805465,0.0005396822,0.00006448653,0.0009409568,0.010456552,0.0013596105,0.011610593,0.13824603,0.8307065,0.00012100015],"about_ca_topic_score_codex":0.004722458,"about_ca_topic_score_gemma":0.0038266587,"teacher_disagreement_score":0.085562505,"about_ca_system_score_codex":0.003137461,"about_ca_system_score_gemma":0.0039865295,"threshold_uncertainty_score":0.2862351},"labels":[],"label_agreement":null},{"id":"W6891916858","doi":"10.48550/arxiv.2308.13129","title":"Accelerating Continuous Integration with Parallel Batch Testing","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Reduction (mathematics); Variable (mathematics); Regression testing; Test case; Test (biology); Batch processing; Execution time; Constant (computer programming); System under test","score_opus":0.19237847088836976,"score_gpt":0.21561095712481182,"score_spread":0.023232486236442057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6891916858","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39229605,0.0013073033,0.5035475,0.00097647845,0.00027987783,0.00073454436,0.00061964244,0.089008495,0.011230053],"genre_scores_gemma":[0.7016498,0.00020259328,0.29167682,0.0002718415,0.00004462841,0.00030015767,0.0012136311,0.002349058,0.0022913958],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9947483,0.0011253481,0.000270266,0.000895591,0.0023618892,0.00059860887],"domain_scores_gemma":[0.9790229,0.007273884,0.0013146969,0.0072810776,0.0042274185,0.0008800179],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043941992,0.0017144228,0.0008497593,0.0013792503,0.00044977586,0.0018302317,0.0039890185,0.0006254889,0.0038995596],"category_scores_gemma":[0.018819962,0.00084939133,0.0009410904,0.0011983607,0.001224003,0.003186813,0.002176341,0.0019941127,0.0016621106],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017241572,0.001538733,0.04214698,0.0006809968,0.00022384152,0.00065298553,0.0005368286,0.19516629,0.1158784,0.007209984,0.017871818,0.61636907],"study_design_scores_gemma":[0.00037935606,0.0016167145,0.013462257,0.00009543807,0.00014056437,0.0003757961,0.00020763114,0.8779366,0.08402837,0.010548312,0.011070245,0.00013874401],"about_ca_topic_score_codex":0.0075104604,"about_ca_topic_score_gemma":0.005309618,"teacher_disagreement_score":0.0075104604,"about_ca_system_score_codex":0.0013837247,"about_ca_system_score_gemma":0.0025549338,"threshold_uncertainty_score":0.023239076},"labels":[],"label_agreement":null},{"id":"W6893218605","doi":"10.5281/zenodo.14950000","title":"Investigating Groundwater Resources using Vertical Electrical Sounding: A Case Study in Baidoa, Somalia.","year":2025,"lang":"en","type":"dissertation","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Situated; Groundwater; Population; Quarter (Canadian coin); Hydrogeology; Bay; Water resources; Drilling","score_opus":0.07027851002359313,"score_gpt":0.31043246749962505,"score_spread":0.2401539574760319,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6893218605","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9933897,0.00045315406,0.0005900692,0.00047140705,0.000012349198,0.00005522558,0.00015291362,0.000008225775,0.004867008],"genre_scores_gemma":[0.9950649,0.0010076084,0.0015675107,0.00010520187,0.000008969976,0.000023553315,0.000074053496,0.0000028284564,0.0021453095],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9997285,0.00009099323,0.000017433016,0.000035478904,0.0000545524,0.00007305426],"domain_scores_gemma":[0.9996954,0.00012805907,0.00006017322,0.000014742298,0.00005354403,0.000048121772],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039228945,0.00037403588,0.0002316058,0.0010843684,0.0018578203,0.0011506304,0.0007195697,0.0013783973,0.001104029],"category_scores_gemma":[0.00065006147,0.00024264057,0.00024910638,0.0022874963,0.0010147875,0.0007064739,0.0008380952,0.0004922649,0.00015359373],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032055436,0.0009857714,0.54934096,0.0015296354,0.00014550469,0.17091839,0.13364376,0.007238124,0.019180113,0.006151833,0.0056548202,0.1048906],"study_design_scores_gemma":[0.00003628346,0.00093511515,0.40397006,0.0005057426,0.00013176897,0.011674268,0.5459181,0.006222038,0.0035118477,0.0017883846,0.025220295,0.00008610343],"about_ca_topic_score_codex":0.03579325,"about_ca_topic_score_gemma":0.13785845,"teacher_disagreement_score":0.03579325,"about_ca_system_score_codex":0.0018710339,"about_ca_system_score_gemma":0.001177681,"threshold_uncertainty_score":0.07116979},"labels":[],"label_agreement":null},{"id":"W6893802102","doi":"10.5281/zenodo.4671169","title":"MANDOLINE: Dynamic Slicing of Android Applications with Trace-Based Alias Analysis","year":2021,"lang":"en","type":"article","venue":"Figshare","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Program slicing; Debugging; Slicing; Android (operating system); Suite; Static analysis; Call graph; Benchmark (surveying)","score_opus":0.018263106225872388,"score_gpt":0.2659056568033319,"score_spread":0.2476425505774595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6893802102","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06378555,0.0009548685,0.8452777,0.00025862877,0.00009767014,0.00033420188,0.0008708941,0.08397871,0.004441807],"genre_scores_gemma":[0.46713778,0.00046329567,0.5218399,0.00018258486,0.000035123252,0.00029877436,0.0020663983,0.005202567,0.002773538],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986533,0.00027244343,0.00014127766,0.00023118977,0.0005679788,0.00013375503],"domain_scores_gemma":[0.99591297,0.0016515278,0.00044310937,0.0012093402,0.0006767661,0.00010639803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010965483,0.0013649013,0.00048150535,0.0019063379,0.00047852006,0.00087229774,0.0016684717,0.0005989761,0.0025069546],"category_scores_gemma":[0.0060956003,0.00061641086,0.0009263963,0.00075463176,0.0009603424,0.0018947546,0.0015343109,0.0011135356,0.0005504943],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009957505,0.00029026132,0.022159081,0.001265918,0.0002963188,0.0013301444,0.0014401725,0.108749256,0.11489629,0.02288701,0.020577557,0.7051122],"study_design_scores_gemma":[0.00010836892,0.00040197908,0.005199033,0.00019602613,0.00013629228,0.0008794017,0.00024708806,0.79555106,0.15382245,0.017048799,0.026281808,0.00012766605],"about_ca_topic_score_codex":0.0060021104,"about_ca_topic_score_gemma":0.0088612605,"teacher_disagreement_score":0.0060021104,"about_ca_system_score_codex":0.000719774,"about_ca_system_score_gemma":0.0015432913,"threshold_uncertainty_score":0.01193434},"labels":[],"label_agreement":null},{"id":"W6894279115","doi":"10.5281/zenodo.7553308","title":"Demystifying Issues, Challenges, and Solutions for Multilingual Software Development (Artifact)","year":2023,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Scripting language; Software; Artifact (error); Process (computing); Coding (social sciences); Sampling (signal processing)","score_opus":0.14318193509934804,"score_gpt":0.304539451893625,"score_spread":0.16135751679427698,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6894279115","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07108642,0.0054311827,0.7874577,0.03810241,0.006314885,0.0017409676,0.007640997,0.011804938,0.07042049],"genre_scores_gemma":[0.24555743,0.0035559644,0.668782,0.0037914936,0.0013661291,0.0012656254,0.010971505,0.0048651695,0.059844717],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99467105,0.0024577486,0.0004129588,0.00058063335,0.0015980512,0.00027956956],"domain_scores_gemma":[0.9858878,0.0061086304,0.0010790392,0.0028924227,0.0032029636,0.0008290902],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068534724,0.0005303756,0.0003105057,0.0032743302,0.0015345716,0.0045197434,0.000750937,0.00074869086,0.010950267],"category_scores_gemma":[0.022887208,0.00030094298,0.00063821475,0.0025811754,0.0014448212,0.004683776,0.005574131,0.001746959,0.0031451778],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015278698,0.0001247501,0.0040721325,0.0011877832,0.000027355836,0.00032446772,0.015482472,0.001005331,0.010843063,0.05467959,0.2032633,0.708837],"study_design_scores_gemma":[0.000028851066,0.00015666116,0.0067374217,0.0010441839,0.000038468104,0.0005891736,0.0068342793,0.005221599,0.012441603,0.03202784,0.9348068,0.00007313803],"about_ca_topic_score_codex":0.0027257719,"about_ca_topic_score_gemma":0.0065314975,"teacher_disagreement_score":0.010950267,"about_ca_system_score_codex":0.0021346554,"about_ca_system_score_gemma":0.00433317,"threshold_uncertainty_score":0.0366323},"labels":[],"label_agreement":null},{"id":"W6902284112","doi":"10.6084/m9.figshare.26653996.v1","title":"Additional file 2 of Performance analysis of conventional and AI-based variant callers using short and long reads","year":2024,"lang":"en","type":"article","venue":"Figshare","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Venn diagram; Degree (music); Indel; Diagram; Training set","score_opus":0.04056998422942117,"score_gpt":0.2757366639307596,"score_spread":0.2351666797013384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6902284112","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005780298,0.000024260096,0.0021638765,0.000100297955,0.00007575235,0.00018025693,0.9937728,0.0021570874,0.00094766147],"genre_scores_gemma":[0.010820855,0.000085237996,0.019920422,0.00066360034,0.00012691754,0.0025326204,0.95058507,0.005363335,0.009901927],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971534,0.00050555967,0.0004338067,0.0008000321,0.000815935,0.00029126485],"domain_scores_gemma":[0.95867383,0.03137416,0.0015106857,0.0032279112,0.004537067,0.0006763147],"candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.004769771,0.0017610949,0.0016187254,0.0034131242,0.0016938541,0.0021026065,0.0031016348,0.0014909485,0.7744319],"category_scores_gemma":[0.04399897,0.0009059358,0.0015565101,0.0043897224,0.0005229137,0.001877713,0.001336523,0.001629827,0.16301426],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00082792714,0.00018666232,0.0042342013,0.0029343723,0.00011038808,0.00013563482,0.00014574533,0.0012166451,0.0012061029,0.0008134191,0.9738511,0.01433784],"study_design_scores_gemma":[0.00565106,0.00081597595,0.04532477,0.0029394764,0.00052288733,0.0010781242,0.00074221485,0.011361324,0.01000845,0.013001851,0.9080341,0.0005196873],"about_ca_topic_score_codex":0.006507228,"about_ca_topic_score_gemma":0.01156156,"teacher_disagreement_score":0.99523026,"about_ca_system_score_codex":0.0013725865,"about_ca_system_score_gemma":0.0025838688,"threshold_uncertainty_score":0.3217455},"labels":[],"label_agreement":null},{"id":"W6910685597","doi":"10.48550/arxiv.1608.07883","title":"Fault Localization in Web Applications via Model Finding","year":2016,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Satisfiability; Set (abstract data type); Focus (optics); Object (grammar); Fault (geology); Boolean satisfiability problem; Web application; Fault model","score_opus":0.07762958019746287,"score_gpt":0.2147679170845426,"score_spread":0.13713833688707971,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6910685597","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010302846,0.00009469971,0.98676515,0.00023036105,0.000011684238,0.000041342606,0.000028309438,0.0018876614,0.00063795055],"genre_scores_gemma":[0.44521287,0.0003196572,0.55157477,0.00017565051,0.000028498616,0.00020841297,0.00019294425,0.0004430402,0.001844229],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969042,0.0010059975,0.0001906738,0.0005833669,0.001059456,0.00025636368],"domain_scores_gemma":[0.9931798,0.004379175,0.0005657955,0.0013559546,0.000438267,0.00008100647],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002169607,0.0009906546,0.00090867496,0.0015435537,0.0007132354,0.0021964984,0.0027014872,0.0017361796,0.0018471879],"category_scores_gemma":[0.012061527,0.00073451234,0.0017360622,0.0011223305,0.0023041242,0.0036989318,0.003416235,0.0023775762,0.00047971634],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023635628,0.00032965885,0.004505964,0.00070406805,0.00018539243,0.0013261833,0.0012108812,0.5081774,0.038205873,0.21344742,0.002982618,0.22868815],"study_design_scores_gemma":[0.00002767731,0.00005682406,0.00016722335,0.00003942008,0.00004941854,0.00027741483,0.00009512709,0.8484577,0.02616852,0.12172335,0.0029145447,0.000022785367],"about_ca_topic_score_codex":0.0023527113,"about_ca_topic_score_gemma":0.002657761,"teacher_disagreement_score":0.0027014872,"about_ca_system_score_codex":0.0013194287,"about_ca_system_score_gemma":0.0016490426,"threshold_uncertainty_score":0.011474133},"labels":[],"label_agreement":null},{"id":"W6912448652","doi":"10.5281/zenodo.4671168","title":"MANDOLINE: Dynamic Slicing of Android Applications with Trace-Based Alias Analysis","year":2021,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Nucleofection; Hyporeflexia; TSG101; Subpoena; Hemopericardium; Tantalum carbide","score_opus":0.020738530750386112,"score_gpt":0.25133448420449733,"score_spread":0.2305959534541112,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6912448652","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0765693,0.0011106383,0.81908065,0.00024734772,0.00010849883,0.0003792598,0.0010186931,0.09683841,0.0046471898],"genre_scores_gemma":[0.488002,0.0005439383,0.49936327,0.00017631653,0.00004170219,0.0003470154,0.0025857636,0.0059425654,0.0029974068],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987079,0.00024239441,0.00014050685,0.00022326395,0.0005525778,0.0001334559],"domain_scores_gemma":[0.9964019,0.0013894706,0.00041528797,0.0010861376,0.00060154527,0.00010574704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000983812,0.0014272212,0.0005005541,0.0018049148,0.00047762686,0.0008689446,0.0016172981,0.0005931129,0.0023407335],"category_scores_gemma":[0.0053097312,0.0006329402,0.00094867597,0.0007124596,0.0009292114,0.0017403151,0.0015061019,0.0010768356,0.00053523976],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012260532,0.00032052334,0.022938447,0.0014667108,0.0003410969,0.0014985204,0.0014269847,0.122090235,0.14094166,0.020355051,0.02154598,0.66584873],"study_design_scores_gemma":[0.00011316437,0.00039436293,0.0053366204,0.00017215087,0.0001403995,0.0007881875,0.00021826322,0.7996261,0.15557595,0.013804176,0.023706865,0.00012376867],"about_ca_topic_score_codex":0.0061832983,"about_ca_topic_score_gemma":0.008462971,"teacher_disagreement_score":0.0061832983,"about_ca_system_score_codex":0.00070575386,"about_ca_system_score_gemma":0.0017052743,"threshold_uncertainty_score":0.0122945905},"labels":[],"label_agreement":null},{"id":"W6931120072","doi":"10.5281/zenodo.4661878","title":"Test Sequence Generation with Cayley Graphs","year":2021,"lang":"en","type":"article","venue":"Figshare","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Test suite; Sequence (biology); Set (abstract data type); Graph; Directed graph; State (computer science)","score_opus":0.08709964338621846,"score_gpt":0.27480716160866375,"score_spread":0.1877075182224453,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6931120072","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010681448,0.000059458183,0.9859949,0.00008741796,0.000012146664,0.0001013427,0.00010783441,0.0014785626,0.00147698],"genre_scores_gemma":[0.21365179,0.00013314982,0.78325987,0.00019129956,0.000021144333,0.00023910649,0.0008064574,0.0004176136,0.0012796136],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99806136,0.0006894876,0.00012577463,0.0003790797,0.0006118048,0.0001325658],"domain_scores_gemma":[0.9956814,0.0027988625,0.0003145443,0.0006416048,0.00048267923,0.00008083182],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012908304,0.00068654853,0.00046841803,0.0013258928,0.00040581045,0.0009035832,0.0014274734,0.000601199,0.003381512],"category_scores_gemma":[0.005950454,0.00045085483,0.0009003124,0.0010416395,0.0010821745,0.001489089,0.000904529,0.0012152508,0.00061333244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002524856,0.00015962351,0.0024274425,0.00040177937,0.000082925624,0.0006166304,0.0002902045,0.4551362,0.038205132,0.19630916,0.0046802782,0.30143815],"study_design_scores_gemma":[0.00005263239,0.00016555365,0.00031484166,0.00004928236,0.000022679953,0.00029172743,0.00004333272,0.8823368,0.025608998,0.08504462,0.0060428395,0.000026619862],"about_ca_topic_score_codex":0.0024235847,"about_ca_topic_score_gemma":0.0028914076,"teacher_disagreement_score":0.003381512,"about_ca_system_score_codex":0.0010766551,"about_ca_system_score_gemma":0.0010903635,"threshold_uncertainty_score":0.011312306},"labels":[],"label_agreement":null},{"id":"W6931820053","doi":"10.5683/sp3/bmr1jt","title":"Supplementary data for: Implementation of Dunaliella tertiolecta and Desmodesmus communis in a photobioreactor prototype for treatment of wastewater in a recirculating aquaculture system","year":2025,"lang":"en","type":"dataset","venue":"Borealis","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Photobioreactor; Wastewater; Aquaculture; Sewage treatment; Recirculating aquaculture system; Dunaliella; Bioprocess","score_opus":0.048858509544572826,"score_gpt":0.34523377894963414,"score_spread":0.2963752694050613,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6931820053","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00023361071,0.000024764839,0.0001928479,0.00003342353,0.00003069219,0.000030368281,0.99864787,0.0004974469,0.0003090427],"genre_scores_gemma":[0.00060727523,0.000024421097,0.0013207941,0.00006163428,0.0000052248724,0.00023987735,0.9970182,0.00018682415,0.00053572486],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983215,0.0002322526,0.000272037,0.0005359596,0.00040996488,0.00022823748],"domain_scores_gemma":[0.9946814,0.0019875364,0.00055599783,0.0010578198,0.0012713004,0.0004459559],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024431625,0.002138245,0.0014521104,0.0030087095,0.0012939157,0.0021869026,0.0027040257,0.0022557578,0.15171103],"category_scores_gemma":[0.008852265,0.00090304104,0.0014374179,0.0037467873,0.0006035874,0.0015106292,0.0021475,0.0023418241,0.094516985],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022720752,0.00010901297,0.003487163,0.0024134896,0.00008315223,0.00005336908,0.0000710203,0.0006977146,0.0012584592,0.00067207305,0.9858483,0.0050790883],"study_design_scores_gemma":[0.00062953704,0.00009454639,0.0142264385,0.000629828,0.00009547696,0.000120342476,0.00017792692,0.0010269424,0.0022466753,0.0019037151,0.9787789,0.000069637324],"about_ca_topic_score_codex":0.012262038,"about_ca_topic_score_gemma":0.029324103,"teacher_disagreement_score":0.15171103,"about_ca_system_score_codex":0.0015438903,"about_ca_system_score_gemma":0.0028556255,"threshold_uncertainty_score":0.50752395},"labels":[],"label_agreement":null},{"id":"W6948256836","doi":"10.48550/arxiv.1403.7261","title":"Generating Complete and Finite Test Suite for ioco: Is It Possible?","year":2014,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Nondeterministic algorithm; Test suite; Conformance testing; Model-based testing; Relation (database); Construct (python library); Domain (mathematical analysis); Test case","score_opus":0.12393388249532612,"score_gpt":0.22448050045350496,"score_spread":0.10054661795817883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6948256836","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049273666,0.00017640932,0.9472348,0.0004014772,0.000026093432,0.00012347072,0.00011356739,0.001261422,0.0013891456],"genre_scores_gemma":[0.5563993,0.00023062248,0.44059384,0.00020228355,0.000056484972,0.00041570977,0.0007629138,0.00044184728,0.00089695747],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99419045,0.0024639063,0.0003576149,0.0007922465,0.0017391408,0.00045678765],"domain_scores_gemma":[0.96405965,0.025936345,0.0014708348,0.006020791,0.0019544084,0.0005579269],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050088535,0.0008170007,0.0011885047,0.0012033839,0.00045441292,0.0013768859,0.0014518906,0.0014015717,0.0018089327],"category_scores_gemma":[0.03332114,0.00045097715,0.0014623328,0.0007245961,0.0024847256,0.0026875164,0.0018000673,0.0015734612,0.0004427071],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009141025,0.0005912881,0.017699812,0.0014690866,0.00028901323,0.0017588394,0.001419172,0.21553172,0.07516966,0.2546347,0.0036574632,0.426865],"study_design_scores_gemma":[0.00017419778,0.00078412215,0.0024133048,0.00027222763,0.00013244168,0.0013877818,0.0002509149,0.6469577,0.067644365,0.2726909,0.007214103,0.000077970944],"about_ca_topic_score_codex":0.0003715365,"about_ca_topic_score_gemma":0.00048292524,"teacher_disagreement_score":0.0050088535,"about_ca_system_score_codex":0.0006180303,"about_ca_system_score_gemma":0.0014634458,"threshold_uncertainty_score":0.026489615},"labels":[],"label_agreement":null},{"id":"W6950093527","doi":"10.5281/zenodo.4673775","title":"Assisting Bug Report Assignment Using Automated Fault Localisation: An Industrial Case Study","year":2021,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Nucleofection; TSG101; Subpoena; Hyporeflexia; Protein isoform; Diafiltration","score_opus":0.1660847211701703,"score_gpt":0.334696775609391,"score_spread":0.16861205443922067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6950093527","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9453923,0.00039531675,0.04927932,0.00045686538,0.000030704625,0.0002834834,0.00039652328,0.0015748829,0.0021906924],"genre_scores_gemma":[0.9469516,0.00016623749,0.05112602,0.00006879141,0.0000137659445,0.00008368525,0.00032766882,0.0001448698,0.0011171838],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99496895,0.0022657546,0.00034020966,0.0006513746,0.0013928802,0.00038093486],"domain_scores_gemma":[0.9623762,0.02573171,0.0027182938,0.0043941233,0.004003154,0.00077647663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042362073,0.0008464786,0.0005364801,0.002193836,0.00090858195,0.001066668,0.0018491414,0.0019176999,0.001499495],"category_scores_gemma":[0.018245444,0.00042999454,0.00054273097,0.0018083742,0.0013767587,0.0010591066,0.001174867,0.0009509596,0.000490674],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022730047,0.0047207708,0.18867083,0.002169623,0.00030084504,0.03835476,0.017502218,0.21904646,0.061594903,0.0060308618,0.010995136,0.44834065],"study_design_scores_gemma":[0.0010378512,0.0058165397,0.12906379,0.0005353432,0.0005164361,0.023287175,0.012371533,0.6445684,0.13275373,0.0073428866,0.042298894,0.00040749996],"about_ca_topic_score_codex":0.006484232,"about_ca_topic_score_gemma":0.008503351,"teacher_disagreement_score":0.006484232,"about_ca_system_score_codex":0.0009971961,"about_ca_system_score_gemma":0.00091076776,"threshold_uncertainty_score":0.022403479},"labels":[],"label_agreement":null},{"id":"W6958025877","doi":"10.6084/m9.figshare.24187913","title":"Additional file 2 of Modeled small airways lung deposition of two fixed-dose triple therapy combinations assessed with in silico functional respiratory imaging","year":2023,"lang":"en","type":"article","venue":"Figshare","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"AstraZeneca (Canada)","funders":"","keywords":"Deposition (geology); In silico; Lung; Respiratory system; Respiratory physiology","score_opus":0.05275886261047619,"score_gpt":0.2692414870036287,"score_spread":0.21648262439315252,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6958025877","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008704433,0.00004719072,0.002734584,0.00018844649,0.00005882094,0.00012624399,0.9905936,0.0034988176,0.0018818815],"genre_scores_gemma":[0.04185903,0.00023617562,0.019604865,0.0008035096,0.00013927362,0.002120437,0.912751,0.007270363,0.015215374],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997296,0.00003955581,0.000032287124,0.00006684946,0.00009903189,0.000032656222],"domain_scores_gemma":[0.99488527,0.003935164,0.0002098721,0.00025104798,0.00063081685,0.00008770557],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00094494486,0.0012473892,0.00092841353,0.0008486786,0.00044824253,0.0012250417,0.0016253794,0.0014642641,0.7171219],"category_scores_gemma":[0.0076322355,0.0005658684,0.00097183255,0.00084290095,0.00020765865,0.0009664237,0.00069855753,0.00082254,0.09527899],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006060226,0.00020185937,0.0030272047,0.0033452895,0.00011127148,0.00021709967,0.000062534666,0.010039861,0.0010757211,0.001444118,0.96533585,0.014533202],"study_design_scores_gemma":[0.006798096,0.0004977477,0.016737306,0.0019843562,0.00042784092,0.0010148564,0.00027913926,0.07144163,0.009470576,0.017248662,0.87382245,0.00027738992],"about_ca_topic_score_codex":0.004969799,"about_ca_topic_score_gemma":0.008032101,"teacher_disagreement_score":0.7171219,"about_ca_system_score_codex":0.00065381866,"about_ca_system_score_gemma":0.00083876785,"threshold_uncertainty_score":0.40349126},"labels":[],"label_agreement":null},{"id":"W6976748706","doi":"10.6068/dp14ba8065bc151","title":"Most Recent Data (2002). Statistics Canada. CANSIM: Society and Community - Time Use | Country: Canada | Table: Health behaviour in school-aged children 2002, Canadian student response to question: How many times a week do you usually eat or drink these items? | Variable: 15 years, French fries, Less than once a week, Males | Units: %, 2002. Data-Planet™ Statistical Ready Reference by Conquest Systems, Inc. Dataset-ID: 075-001-191.","year":2015,"lang":"en","type":"other","venue":"Data Planet","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Census; Official statistics; Socioeconomic status; Population; Population statistics; Economic statistics; Descriptive statistics; Social statistics; Statistics education","score_opus":0.03957358349309021,"score_gpt":0.2749211646409417,"score_spread":0.23534758114785148,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6976748706","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00004205098,0.00002752245,0.000011747392,0.00009111121,0.000021624839,0.000011606844,0.99928206,0.000037242313,0.00047504785],"genre_scores_gemma":[0.0007153554,0.00018080912,0.00023458431,0.00013321455,0.00001747903,0.00015812273,0.99569976,0.00007957525,0.002781156],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9951696,0.00032000593,0.00059302984,0.0005488695,0.0022583236,0.0011102154],"domain_scores_gemma":[0.9524662,0.0020317123,0.0012384758,0.0012726766,0.040873975,0.002116992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026991013,0.0027627072,0.0032196427,0.0073451516,0.003708338,0.005065746,0.005825215,0.00183944,0.11021641],"category_scores_gemma":[0.024557069,0.0019642473,0.002832523,0.04171904,0.00071539433,0.0024319654,0.0026219983,0.0036261894,0.06395958],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002521066,0.000008714987,0.0010932155,0.00028630625,0.000018426947,0.000005127638,0.000025726951,0.00007938793,0.0000065990157,0.0001646824,0.9972523,0.0010343499],"study_design_scores_gemma":[0.0003772772,0.000022912729,0.053748842,0.0014192667,0.00011278396,0.000029314295,0.0008748716,0.00045194625,0.00018371336,0.0005449891,0.9420893,0.00014479233],"about_ca_topic_score_codex":0.9945022,"about_ca_topic_score_gemma":0.9923044,"teacher_disagreement_score":0.11021641,"about_ca_system_score_codex":0.049272772,"about_ca_system_score_gemma":0.11781506,"threshold_uncertainty_score":0.36871064},"labels":[],"label_agreement":null},{"id":"W6977087287","doi":"10.6084/m9.figshare.26653996","title":"Additional file 2 of Performance analysis of conventional and AI-based variant callers using short and long reads","year":2024,"lang":"en","type":"article","venue":"Figshare","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Venn diagram; Degree (music); Indel; Diagram; Training set","score_opus":0.04056998422942117,"score_gpt":0.2757366639307596,"score_spread":0.2351666797013384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6977087287","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005780298,0.000024260096,0.0021638765,0.000100297955,0.00007575235,0.00018025693,0.9937728,0.0021570874,0.00094766147],"genre_scores_gemma":[0.010820855,0.000085237996,0.019920422,0.00066360034,0.00012691754,0.0025326204,0.95058507,0.005363335,0.009901927],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971534,0.00050555967,0.0004338067,0.0008000321,0.000815935,0.00029126485],"domain_scores_gemma":[0.95867383,0.03137416,0.0015106857,0.0032279112,0.004537067,0.0006763147],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.004769771,0.0017610949,0.0016187254,0.0034131242,0.0016938541,0.0021026065,0.0031016348,0.0014909485,0.7744319],"category_scores_gemma":[0.04399897,0.0009059358,0.0015565101,0.0043897224,0.0005229137,0.001877713,0.001336523,0.001629827,0.16301426],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00082792714,0.00018666232,0.0042342013,0.0029343723,0.00011038808,0.00013563482,0.00014574533,0.0012166451,0.0012061029,0.0008134191,0.9738511,0.01433784],"study_design_scores_gemma":[0.00565106,0.00081597595,0.04532477,0.0029394764,0.00052288733,0.0010781242,0.00074221485,0.011361324,0.01000845,0.013001851,0.9080341,0.0005196873],"about_ca_topic_score_codex":0.006507228,"about_ca_topic_score_gemma":0.01156156,"teacher_disagreement_score":0.7744319,"about_ca_system_score_codex":0.0013725865,"about_ca_system_score_gemma":0.0025838688,"threshold_uncertainty_score":0.3217455},"labels":[],"label_agreement":null},{"id":"W6979226034","doi":"","title":"Exploring Challenges in Test Mocking: Developer Questions and Insights from StackOverflow","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; University of Saskatchewan","keywords":"Latent Dirichlet allocation; Popularity; Key (lock); Selection (genetic algorithm); Advice (programming); Test (biology); Topic model; Unit testing","score_opus":0.20006299142686132,"score_gpt":0.28784583702743455,"score_spread":0.08778284560057323,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6979226034","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.931797,0.0015035284,0.047277436,0.010830687,0.00012290737,0.00014961127,0.0002582779,0.000408663,0.0076519414],"genre_scores_gemma":[0.9837187,0.00047277063,0.01252557,0.00095554313,0.000070634786,0.00010639171,0.00023575596,0.00024690878,0.0016676415],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9654119,0.02365401,0.0015411507,0.0027272508,0.005111669,0.001554022],"domain_scores_gemma":[0.7122222,0.23925914,0.017258056,0.007914318,0.018401762,0.004944553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.045989905,0.0007984728,0.00059006474,0.0046285405,0.0035028541,0.0060838596,0.0015568896,0.002480053,0.0011960189],"category_scores_gemma":[0.1731584,0.0008303121,0.0006199834,0.0022084487,0.004354738,0.01131241,0.0053894063,0.0027518992,0.00044868948],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018613861,0.0001761737,0.15354876,0.0005435928,0.000050560655,0.0020205437,0.7187422,0.001609978,0.0055205706,0.007668242,0.0065965927,0.103336744],"study_design_scores_gemma":[0.00006149926,0.0005630139,0.13209519,0.0013809716,0.00011586737,0.0032215551,0.67994463,0.024180165,0.007441755,0.027504029,0.12310068,0.00039058278],"about_ca_topic_score_codex":0.0035386395,"about_ca_topic_score_gemma":0.0068081995,"teacher_disagreement_score":0.045989905,"about_ca_system_score_codex":0.0036622013,"about_ca_system_score_gemma":0.0029828986,"threshold_uncertainty_score":0.24322075},"labels":[],"label_agreement":null},{"id":"W6979250945","doi":"","title":"Advanced Penetration Testing for Enhancing 5G Security","year":2024,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Limiting; Hyporeflexia; Proteogenomics; TSG101; Nucleofection","score_opus":0.06575916532810426,"score_gpt":0.21057147694091943,"score_spread":0.14481231161281516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6979250945","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13672951,0.012219828,0.767583,0.004381379,0.00049312937,0.0006833554,0.00026753143,0.0031287975,0.07451347],"genre_scores_gemma":[0.83575755,0.008714791,0.14702953,0.0010935232,0.00010968027,0.00029147996,0.00025557767,0.0003531347,0.006394746],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9951649,0.0018633243,0.00015652702,0.00026687948,0.002224631,0.00032379042],"domain_scores_gemma":[0.9897423,0.0061522475,0.000918146,0.0010917004,0.0019013037,0.00019426171],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038482647,0.0009312547,0.0005040474,0.0022854742,0.0005118162,0.0015748844,0.0012903209,0.0011742931,0.0043186108],"category_scores_gemma":[0.011875084,0.0003156845,0.00056832697,0.0012132105,0.0014828212,0.0035291805,0.001627381,0.0013926399,0.0009029816],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019723702,0.0003109835,0.015533787,0.0012182106,0.00015384408,0.0014424486,0.001786059,0.055166546,0.0507831,0.10051546,0.0100555355,0.7628368],"study_design_scores_gemma":[0.0001397759,0.0032792613,0.018766638,0.0035086202,0.00042241765,0.0071465313,0.0029490322,0.37610802,0.1772112,0.16057001,0.24956995,0.00032845352],"about_ca_topic_score_codex":0.0015471383,"about_ca_topic_score_gemma":0.0015705797,"teacher_disagreement_score":0.0043186108,"about_ca_system_score_codex":0.00103637,"about_ca_system_score_gemma":0.0009133984,"threshold_uncertainty_score":0.020351827},"labels":[],"label_agreement":null},{"id":"W6979280317","doi":"","title":"Simulink Mutation Testing using CodeBERT","year":2025,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Mutation; Mutation testing; Process (computing); Mutant; Frameshift mutation","score_opus":0.11593985627887145,"score_gpt":0.2193839124784322,"score_spread":0.10344405619956075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6979280317","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07619535,0.00022050757,0.865149,0.00022590216,0.000089517926,0.00012531856,0.00040353296,0.050524484,0.00706633],"genre_scores_gemma":[0.58229256,0.00017306973,0.40847987,0.0001721387,0.00001965384,0.00012829438,0.0008319287,0.004021105,0.0038813686],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9984896,0.0003726916,0.000080810336,0.00021748304,0.0007300407,0.00010935974],"domain_scores_gemma":[0.9950813,0.0026923863,0.00054031186,0.00097001274,0.0006181791,0.00009782207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011901463,0.0011847229,0.000442323,0.0010760106,0.00034176683,0.0007221985,0.0016354745,0.0008601066,0.002964979],"category_scores_gemma":[0.007724332,0.00044888433,0.0005870087,0.00039028173,0.0010194819,0.0011629366,0.0010061944,0.0008626152,0.00067031157],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008872078,0.0003792598,0.010603551,0.0006357002,0.00020848747,0.0014661452,0.00063895073,0.5734665,0.09957733,0.05970432,0.0107697705,0.2416628],"study_design_scores_gemma":[0.000048793445,0.00018962537,0.0005616753,0.00006261726,0.0000393629,0.00042143874,0.000039359562,0.8958585,0.082541145,0.008041873,0.012156555,0.00003903353],"about_ca_topic_score_codex":0.0033816916,"about_ca_topic_score_gemma":0.0045595467,"teacher_disagreement_score":0.0033816916,"about_ca_system_score_codex":0.0007895371,"about_ca_system_score_gemma":0.0012280477,"threshold_uncertainty_score":0.009918809},"labels":[],"label_agreement":null},{"id":"W6979735060","doi":"","title":"Acoustical comparisons of Alberta concert halls","year":2003,"lang":"en","type":"article","venue":"NPARC","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"","score_opus":0.028814005696070427,"score_gpt":0.27822762024950576,"score_spread":0.24941361455343533,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6979735060","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9365849,0.00010633035,0.0007291277,0.00018961636,0.000100858946,0.000029795428,0.00075372367,0.000098096614,0.061407577],"genre_scores_gemma":[0.98993075,0.000048565038,0.00036046014,0.000037127742,0.000026445758,0.000012687846,0.0005404039,0.00004055463,0.009003034],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99888974,0.00012438816,0.000015557605,0.00012832224,0.0005586061,0.00028332885],"domain_scores_gemma":[0.9979972,0.0002609811,0.00006009094,0.000051697203,0.001317113,0.0003129149],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006564201,0.0003080877,0.00035437706,0.0021014672,0.002713661,0.0020128482,0.00096081366,0.0006783598,0.009011388],"category_scores_gemma":[0.0020078877,0.00022271113,0.00019237693,0.0026026815,0.0009008684,0.00040553397,0.0012033066,0.00061434606,0.0013851521],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.014794057,0.0014724467,0.49538794,0.00039267997,0.00036453677,0.0033667826,0.043601733,0.020770788,0.07853161,0.0132060535,0.047790788,0.28032064],"study_design_scores_gemma":[0.000052224004,0.00018078255,0.953639,0.000043690838,0.000051815914,0.00008627643,0.018258844,0.003303753,0.002553075,0.00029893225,0.021482432,0.000049342572],"about_ca_topic_score_codex":0.537305,"about_ca_topic_score_gemma":0.81122464,"teacher_disagreement_score":0.462695,"about_ca_system_score_codex":0.0053057736,"about_ca_system_score_gemma":0.0027015964,"threshold_uncertainty_score":0.93083984},"labels":[],"label_agreement":null},{"id":"W6990863403","doi":"","title":"Enabling Language-Specific Transformations in Language-Agnostic Program Reduction","year":2023,"lang":"en","type":"dissertation","venue":"UWSpace (University of Waterloo)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Blackberry (Canada)","funders":"","keywords":"Metis; Program transformation; Transformation (genetics); Reduction (mathematics); Matching (statistics); Domain (mathematical analysis); Abstract syntax tree; Rewriting","score_opus":0.015273204512537096,"score_gpt":0.24332602162394068,"score_spread":0.22805281711140357,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6990863403","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022011189,0.0005920455,0.9212131,0.00044402105,0.000086339955,0.00047616812,0.0004525646,0.048121743,0.006602843],"genre_scores_gemma":[0.17153607,0.0005459236,0.8133913,0.0006412807,0.000069646325,0.00048203598,0.0017440513,0.0069283033,0.004661299],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995464,0.0009853175,0.0004765643,0.0008745007,0.0017269113,0.00047272805],"domain_scores_gemma":[0.99461776,0.0016428236,0.00055560266,0.0021910288,0.0008710969,0.00012166394],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027494272,0.0016465378,0.0007326706,0.001719207,0.0007087552,0.001893178,0.0028992929,0.0012890587,0.002580299],"category_scores_gemma":[0.0076255132,0.0008159022,0.0018866325,0.0014006003,0.0024807765,0.00441226,0.0034851776,0.0033641895,0.0014839091],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004445397,0.00078315026,0.007846039,0.0020002332,0.00021322517,0.0010749375,0.002019872,0.06701075,0.17750394,0.14192471,0.029926213,0.56925243],"study_design_scores_gemma":[0.00016665575,0.00058174034,0.0032576614,0.0003221869,0.00029432244,0.001639478,0.00047932973,0.44277883,0.26327774,0.0986117,0.18834181,0.00024854948],"about_ca_topic_score_codex":0.0034005516,"about_ca_topic_score_gemma":0.005906434,"teacher_disagreement_score":0.0034005516,"about_ca_system_score_codex":0.0012946577,"about_ca_system_score_gemma":0.003244876,"threshold_uncertainty_score":0.014540553},"labels":[],"label_agreement":null},{"id":"W6998781786","doi":"","title":"Automatic fault localization in concurrent programs using noising and search strategies","year":2021,"lang":"en","type":"dissertation","venue":"e-scholar@UOIT (University of Ontario Institute of Technology)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Debugging; Concurrency; Program slicing; Set (abstract data type); Java; Code (set theory); Heuristic; Fault (geology); Algorithmic program debugging; Software","score_opus":0.027375496572537194,"score_gpt":0.2635435827122169,"score_spread":0.2361680861396797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6998781786","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22091013,0.00023405519,0.76986426,0.0001711524,0.000019489398,0.0001358828,0.000076014854,0.0066477195,0.0019412547],"genre_scores_gemma":[0.63558847,0.00008704978,0.3621807,0.000089542955,0.0000068534036,0.000102913036,0.0001756513,0.00042264862,0.001346267],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99876,0.00036049788,0.00007792411,0.00024245861,0.00041489018,0.00014431712],"domain_scores_gemma":[0.9927124,0.005005729,0.0007338222,0.0006839948,0.00072346086,0.00014067661],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015712072,0.0006503549,0.0006029426,0.0011946891,0.0006918105,0.00086834707,0.0012373491,0.0006556335,0.0012460853],"category_scores_gemma":[0.0078907935,0.00048522378,0.00056967087,0.00081440003,0.0010761551,0.0014090696,0.0009393552,0.0007213562,0.00021875976],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007314388,0.00054152776,0.022859022,0.00047210848,0.00010231384,0.0004378403,0.0012000538,0.35569403,0.14805752,0.019265668,0.0020801786,0.4485583],"study_design_scores_gemma":[0.00007591152,0.00021542309,0.002143789,0.000031145144,0.000038627284,0.00015087008,0.00017856306,0.9401344,0.047037963,0.008490688,0.0014665542,0.000036024714],"about_ca_topic_score_codex":0.007884809,"about_ca_topic_score_gemma":0.016042471,"teacher_disagreement_score":0.007884809,"about_ca_system_score_codex":0.0010122282,"about_ca_system_score_gemma":0.0024180596,"threshold_uncertainty_score":0.01567781},"labels":[],"label_agreement":null},{"id":"W7000725270","doi":"","title":"Fuzzing OpenMP Compilers","year":2024,"lang":"en","type":"dissertation","venue":"UWSpace (University of Waterloo)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Blackberry (Canada)","funders":"","keywords":"Fuzz testing; Compiler; Correctness; Test suite; Suite; Flexibility (engineering); Compile time; Code (set theory)","score_opus":0.015617168249005255,"score_gpt":0.22445262208487107,"score_spread":0.20883545383586583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7000725270","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42246282,0.000857695,0.5377343,0.00062037347,0.00022338364,0.00045214553,0.00065113645,0.024991082,0.012006972],"genre_scores_gemma":[0.7262036,0.00037445285,0.2646458,0.00030757117,0.000035356905,0.00029839238,0.0015223685,0.002345344,0.00426714],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967085,0.00081425766,0.00019717836,0.0004028975,0.0016031063,0.00027397394],"domain_scores_gemma":[0.9915833,0.004479501,0.0005992434,0.001704364,0.0015168573,0.000116879615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020654604,0.0006938362,0.0004165918,0.0009896806,0.0005566982,0.0011928976,0.0013823365,0.0007284199,0.0017747807],"category_scores_gemma":[0.012984166,0.0005532992,0.0006265309,0.0005457501,0.00094632886,0.001434294,0.0012557644,0.0010661148,0.0005305796],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011589428,0.0005081566,0.020799415,0.0011727797,0.0001815995,0.0015160398,0.0015680338,0.21194164,0.16926399,0.052837953,0.015348272,0.52370316],"study_design_scores_gemma":[0.0000977387,0.00037620857,0.0042432263,0.00030725723,0.00013094422,0.000851011,0.00021728227,0.67072165,0.26936528,0.023123851,0.030476036,0.00008955111],"about_ca_topic_score_codex":0.0016418985,"about_ca_topic_score_gemma":0.0018579918,"teacher_disagreement_score":0.0020654604,"about_ca_system_score_codex":0.00083350274,"about_ca_system_score_gemma":0.0013769096,"threshold_uncertainty_score":0.010923326},"labels":[],"label_agreement":null},{"id":"W7008673472","doi":"","title":"Complemento para la automatización de pruebas de aceptación para Visual Studio","year":2011,"lang":"es","type":"other","venue":"e-Archivo (Carlos III University of Madrid)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Studio; Software","score_opus":0.031688985633827604,"score_gpt":0.27727159360236614,"score_spread":0.24558260796853854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7008673472","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0060420707,0.0005885126,0.8755552,0.00031561733,0.00025129932,0.00031717384,0.0005531689,0.0984583,0.017918622],"genre_scores_gemma":[0.11718838,0.0009726324,0.8106566,0.0006516762,0.00013285378,0.0005568981,0.0021236395,0.015371341,0.05234608],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9965262,0.0006603249,0.00023974341,0.0007767391,0.0014676936,0.0003292884],"domain_scores_gemma":[0.9942468,0.0019920075,0.00025701264,0.001870027,0.0013347407,0.0002995452],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025023785,0.0025057576,0.0013690797,0.0021205654,0.0008190949,0.0049066744,0.002892916,0.0019111417,0.048974752],"category_scores_gemma":[0.009307555,0.0014430144,0.0018043057,0.0011818745,0.0011293036,0.003791238,0.0031363205,0.0032588465,0.022411313],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017792133,0.00034363425,0.001997394,0.0015614564,0.00015828127,0.0005587912,0.0017080014,0.008248406,0.08073432,0.022355389,0.042422954,0.8381322],"study_design_scores_gemma":[0.00046088695,0.0007936247,0.0038144318,0.00078216096,0.00031317392,0.0026357358,0.00069217384,0.25210217,0.25343975,0.02808017,0.4564788,0.00040687504],"about_ca_topic_score_codex":0.005749638,"about_ca_topic_score_gemma":0.005747412,"teacher_disagreement_score":0.048974752,"about_ca_system_score_codex":0.0011088923,"about_ca_system_score_gemma":0.0017617699,"threshold_uncertainty_score":0.1638369},"labels":[],"label_agreement":null},{"id":"W7018138705","doi":"","title":"Darker Politics: Democracies, Labour Rights and Climate Change - Poster","year":2022,"lang":"en","type":"other","venue":"York University Digital Library (York University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Feud; Journalism; Climate change; Event (particle physics); Spanish Civil War; Civil rights; Newspaper","score_opus":0.01294080700003063,"score_gpt":0.1718596429534687,"score_spread":0.15891883595343806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7018138705","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013395763,0.0071973926,0.0007771098,0.037759375,0.0044674356,0.000039863196,0.000836341,0.00013364456,0.9353932],"genre_scores_gemma":[0.2730817,0.008583223,0.0010555481,0.0055824914,0.003391097,0.00008101813,0.0011173403,0.0003441821,0.7067634],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9997887,0.00008467593,0.0000032617456,0.00002450372,0.000045596847,0.000053231663],"domain_scores_gemma":[0.9996296,0.00015021436,0.000017126418,0.000030121504,0.000035540852,0.00013736374],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078308384,0.00029625636,0.00015053382,0.00047168307,0.0029229438,0.004044125,0.00029997248,0.0009156064,0.13402824],"category_scores_gemma":[0.0011300569,0.00014074909,0.0002150958,0.00090118265,0.0016225877,0.002008081,0.0019699258,0.0016106733,0.0077975513],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000079431295,0.00006422306,0.0010834232,0.000119239136,0.0000069245434,0.00016806612,0.003810492,0.00016955235,0.00022125944,0.13177092,0.8159416,0.046564836],"study_design_scores_gemma":[0.000013836458,0.00002104306,0.0039490866,0.00020808524,0.000006352748,0.000056324996,0.006907098,0.0002174387,0.00023083537,0.032281473,0.9560962,0.000012195757],"about_ca_topic_score_codex":0.011883715,"about_ca_topic_score_gemma":0.05127474,"teacher_disagreement_score":0.13402824,"about_ca_system_score_codex":0.0025938277,"about_ca_system_score_gemma":0.000959239,"threshold_uncertainty_score":0.44836915},"labels":[],"label_agreement":null},{"id":"W7033563533","doi":"","title":"Real-time detection of water quality aberrations in a water distribution system","year":2009,"lang":"en","type":"dissertation","venue":"The Atrium (University of Guelph)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Contamination; Water quality; Tap water; Turbidity; Water pollution; Sewage; Water treatment","score_opus":0.014269534646226862,"score_gpt":0.2367184476112001,"score_spread":0.22244891296497324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7033563533","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9715835,0.00028680958,0.025132185,0.00014887177,0.00004064618,0.000080889484,0.00026143112,0.0010193336,0.0014463977],"genre_scores_gemma":[0.98451173,0.00014082814,0.013535963,0.00005178555,0.000008247971,0.00002791115,0.00013591585,0.000020546779,0.0015671467],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995914,0.000039528022,0.000012398334,0.00009899601,0.00021106927,0.000046679528],"domain_scores_gemma":[0.9996068,0.000101459504,0.00011147898,0.000024468161,0.00012574182,0.00003002265],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002896448,0.0002532633,0.00031058773,0.0004965125,0.00022365144,0.0004802638,0.0003921702,0.00045137084,0.0008408088],"category_scores_gemma":[0.0005620158,0.00011992142,0.00013068467,0.00041424425,0.00022189232,0.00033237287,0.00028142784,0.00033402024,0.00024124037],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069662486,0.00012318847,0.017608011,0.00014198445,0.000016628428,0.00018703131,0.0003178083,0.0027716944,0.9371919,0.00013842648,0.00045087896,0.040355813],"study_design_scores_gemma":[0.000033018634,0.0011587858,0.048104443,0.00001813359,0.00003500508,0.00026283224,0.0003125784,0.061538603,0.88542485,0.00016963787,0.0028979469,0.00004419507],"about_ca_topic_score_codex":0.0014437663,"about_ca_topic_score_gemma":0.0016405345,"teacher_disagreement_score":0.0014437663,"about_ca_system_score_codex":0.00069251365,"about_ca_system_score_gemma":0.00023943104,"threshold_uncertainty_score":0.005024612},"labels":[],"label_agreement":null},{"id":"W7034311940","doi":"","title":"Superkül Completes Canada’s First Active House – Azure Magazine","year":2013,"lang":"en","type":"other","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Term (time); Work (physics); Doors; Order (exchange)","score_opus":0.013561359358344214,"score_gpt":0.20949878691136892,"score_spread":0.1959374275530247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7034311940","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033661616,0.0032104673,0.0025261652,0.025482705,0.00889375,0.00009645566,0.001691929,0.002381728,0.95235074],"genre_scores_gemma":[0.0036829836,0.00048334402,0.0007181453,0.0006274336,0.00030509382,0.00000760698,0.00032515937,0.00030702844,0.99354315],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982503,0.00006734963,0.000022883089,0.0001437694,0.0012262005,0.00028947234],"domain_scores_gemma":[0.99645895,0.00017254084,0.000039356415,0.00020719299,0.002146166,0.0009759045],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0012451179,0.00092987786,0.00045266436,0.0017839608,0.0054726577,0.007610377,0.0011559206,0.0020947303,0.34553385],"category_scores_gemma":[0.002418873,0.00045831728,0.00051867735,0.0011375037,0.0011724483,0.0017053393,0.0016194176,0.0024121143,0.10903031],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019403786,0.000017299792,0.00015711394,0.00001533235,0.0000029767596,0.000044343506,0.0000351044,0.000072201976,0.00024391276,0.005463432,0.9655476,0.02838121],"study_design_scores_gemma":[0.0000039265424,0.00000495323,0.00045881837,0.000013371797,0.0000020303041,0.000016575766,0.000057874764,0.00012019098,0.00015662202,0.0004476604,0.99871314,0.0000047456824],"about_ca_topic_score_codex":0.59392464,"about_ca_topic_score_gemma":0.8829928,"teacher_disagreement_score":0.40607536,"about_ca_system_score_codex":0.012610799,"about_ca_system_score_gemma":0.020785546,"threshold_uncertainty_score":0.9335165},"labels":[],"label_agreement":null},{"id":"W7036048626","doi":"","title":"APG 504 - Floating Heads","year":2022,"lang":"en","type":"other","venue":"Bulletin of Miscellaneous Information (Royal Gardens Kew)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Crew; Aviation; Certificate; Crash; Aviation safety","score_opus":0.007905021846431473,"score_gpt":0.1957923208804181,"score_spread":0.18788729903398663,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036048626","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005608416,0.000393779,0.0042519085,0.0017526123,0.0021451935,0.00019187137,0.0054488955,0.015810862,0.96944404],"genre_scores_gemma":[0.0032429341,0.00023131479,0.00067352917,0.0011279745,0.00032736547,0.00009934172,0.002931249,0.0012657859,0.99010044],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995515,0.00005032242,0.000022597604,0.000107501604,0.0001850283,0.000082957195],"domain_scores_gemma":[0.9988254,0.00012320971,0.000042108273,0.00022122319,0.0005401194,0.00024794045],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00061847403,0.0011843527,0.00056087214,0.00091209554,0.0015107004,0.003061763,0.0015870209,0.0031472852,0.9081473],"category_scores_gemma":[0.0020853232,0.0005352928,0.00051937514,0.00078515627,0.00042649682,0.002704374,0.0028490885,0.0018374188,0.8822323],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000052623396,0.000011773106,0.00005211722,0.00003397914,0.0000016830576,0.00006567563,0.000018363,0.000024216124,0.00040657117,0.00074210076,0.9608423,0.037748612],"study_design_scores_gemma":[0.000026781445,0.000026640118,0.00020162757,0.00003335501,0.0000026924329,0.00015590053,0.000035032772,0.00010373624,0.00032966738,0.0003998683,0.99867755,0.000007058471],"about_ca_topic_score_codex":0.0029775682,"about_ca_topic_score_gemma":0.0035526643,"teacher_disagreement_score":0.9081473,"about_ca_system_score_codex":0.0006210639,"about_ca_system_score_gemma":0.00072804245,"threshold_uncertainty_score":0.13101679},"labels":[],"label_agreement":null},{"id":"W7036703262","doi":"","title":"Conference report: opportunity and challenge in wood drying: Quality control and energy saving","year":2014,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quality (philosophy); Theme (computing); Control (management); European union; Energy (signal processing)","score_opus":0.05657642606882203,"score_gpt":0.29346110468898073,"score_spread":0.23688467862015872,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036703262","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023418613,0.12019753,0.017005293,0.18337448,0.29788592,0.0019547353,0.006886708,0.0015228541,0.34775388],"genre_scores_gemma":[0.097900435,0.077669844,0.010768219,0.015744057,0.07000218,0.0012382471,0.009033953,0.0012743529,0.7163689],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9960284,0.00056531775,0.0002047822,0.00047684315,0.0019533664,0.00077127205],"domain_scores_gemma":[0.99322283,0.0009858034,0.00029004342,0.00027762793,0.00334628,0.0018773954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055521145,0.0016960552,0.0006826538,0.0020995513,0.0026159897,0.0071012513,0.0017661891,0.004787877,0.06525235],"category_scores_gemma":[0.0054025296,0.000394775,0.0010188804,0.0025709646,0.00075929385,0.004400608,0.0029556074,0.0047476753,0.018024534],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002321525,0.00023560537,0.0008980376,0.0007228615,0.000032635395,0.0002776107,0.00027859016,0.00042180752,0.002049565,0.0040100208,0.9184933,0.07234791],"study_design_scores_gemma":[0.000030399951,0.00031012713,0.0020308215,0.00031468875,0.000036452817,0.0002863124,0.00077503256,0.00042297225,0.0027890583,0.0015614479,0.9914048,0.000037959522],"about_ca_topic_score_codex":0.0030864798,"about_ca_topic_score_gemma":0.0044612647,"teacher_disagreement_score":0.06525235,"about_ca_system_score_codex":0.0021045643,"about_ca_system_score_gemma":0.0046480163,"threshold_uncertainty_score":0.21829087},"labels":[],"label_agreement":null},{"id":"W7036737297","doi":"","title":"Cost effective basement wall drainage alternatives employing exterior insulation basement systems (EIBS)","year":2001,"lang":"en","type":"article","venue":"NPARC","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"National Research Council Canada","keywords":"Drainage; Basement; Thermal insulation; Moisture; Current (fluid)","score_opus":0.034961439353279875,"score_gpt":0.29583459653355165,"score_spread":0.26087315718027176,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036737297","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98445106,0.0004802974,0.010645163,0.000040105424,0.000016359952,0.000038470702,0.00006821446,0.00008323702,0.004177214],"genre_scores_gemma":[0.98893076,0.0002632886,0.009818015,0.000008175905,0.0000046496943,0.000010924369,0.00006728983,0.000015170689,0.00088179496],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995408,0.00014523313,0.000029637376,0.000036942078,0.00017574534,0.00007161416],"domain_scores_gemma":[0.99942636,0.00017898442,0.00016567679,0.00005922161,0.000110146764,0.000059574286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048918166,0.0004487487,0.00032896915,0.0010629852,0.00022813985,0.000706061,0.00037700916,0.0002141369,0.0019313148],"category_scores_gemma":[0.0014490216,0.00013232356,0.000241107,0.0008685937,0.00023927813,0.00069906464,0.0005596837,0.00020489648,0.00028535028],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028223477,0.0011011257,0.042189598,0.0013246344,0.00015026731,0.0012878731,0.00035870986,0.103295006,0.3609171,0.006461672,0.0009936932,0.479098],"study_design_scores_gemma":[0.0004374183,0.016171169,0.083380446,0.00031146366,0.00057264406,0.002518638,0.0024616786,0.1292803,0.7267787,0.005347774,0.032611005,0.00012873804],"about_ca_topic_score_codex":0.00039780216,"about_ca_topic_score_gemma":0.0015375912,"teacher_disagreement_score":0.0019313148,"about_ca_system_score_codex":0.0002724176,"about_ca_system_score_gemma":0.00022297415,"threshold_uncertainty_score":0.0064608455},"labels":[],"label_agreement":null},{"id":"W7036862229","doi":"","title":"2003 Christmas Bird Counts in Nebraska","year":2004,"lang":"en","type":"article","venue":"Lincoln (University of Nebraska)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Snow; Buteo; Waterfowl; Dozen; Ornithology","score_opus":0.009654367088606739,"score_gpt":0.19573137451864106,"score_spread":0.18607700743003433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036862229","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9790222,0.00043219922,0.00038591158,0.00032737837,0.00010918322,0.00004049851,0.0034946695,0.00021074478,0.015977168],"genre_scores_gemma":[0.98055035,0.0004541132,0.0015144147,0.00016279044,0.000046412002,0.000035705827,0.003958869,0.00004773809,0.013229635],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9995925,0.000029318675,0.000033428547,0.00008906241,0.00017157734,0.00008405466],"domain_scores_gemma":[0.99867475,0.000101003236,0.0003133343,0.000048504095,0.000498066,0.00036435915],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00023126599,0.00015962035,0.0001253123,0.0013323678,0.0012459417,0.00057265384,0.0005947545,0.00019095003,0.00540035],"category_scores_gemma":[0.0011219403,0.00020759093,0.00007309365,0.0009097101,0.00016240234,0.00038006384,0.00044075708,0.00034200933,0.00081047113],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019187825,0.00014633343,0.89065564,0.00012620843,0.000024949253,0.0007512696,0.0026655884,0.00014596965,0.0048675165,0.00017421664,0.01938922,0.08086113],"study_design_scores_gemma":[0.0000043932855,0.0000623567,0.98215,0.000023680286,0.0000052253567,0.0003385702,0.0013976161,0.000309208,0.0006249152,0.000029945708,0.015046818,0.0000071799273],"about_ca_topic_score_codex":0.25822794,"about_ca_topic_score_gemma":0.66843337,"teacher_disagreement_score":0.25822794,"about_ca_system_score_codex":0.0018559332,"about_ca_system_score_gemma":0.0011676536,"threshold_uncertainty_score":0.5134498},"labels":[],"label_agreement":null},{"id":"W7038428449","doi":"","title":"Haçlıların Anadolu’da kuşattığı kaleler ve Türk savunması","year":2016,"lang":"tr","type":"dissertation","venue":"DSpace - AKÜ (Afyon Kocatepe University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Field (mathematics); Point (geometry); Quarter (Canadian coin)","score_opus":0.0204608054640158,"score_gpt":0.2341937781290641,"score_spread":0.21373297266504832,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7038428449","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3596681,0.022204632,0.025186697,0.013746966,0.0024639172,0.00056014775,0.0016212295,0.0012040063,0.57334423],"genre_scores_gemma":[0.73532754,0.011775711,0.020586083,0.0022430369,0.00029510196,0.00028581242,0.0014389716,0.0003690682,0.22767851],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986557,0.00024382239,0.000082454826,0.00024797427,0.00046703243,0.00030295926],"domain_scores_gemma":[0.99865663,0.00017385794,0.00019578903,0.000117026764,0.00066022045,0.00019639813],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010795395,0.0012355096,0.0006588902,0.0013312262,0.0043581943,0.005272671,0.000992069,0.0021344665,0.039241903],"category_scores_gemma":[0.0020872352,0.00052209967,0.00081653456,0.001659921,0.0024306967,0.0031802289,0.0036716734,0.0019071827,0.011008316],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013967312,0.00059640163,0.061700705,0.0041358485,0.00025827775,0.005951957,0.030429555,0.0020390896,0.037156235,0.09305465,0.07803401,0.68524647],"study_design_scores_gemma":[0.00004869434,0.0003854687,0.04253262,0.001197723,0.0001049228,0.0023807813,0.037613362,0.0011204324,0.010648123,0.008348945,0.89546216,0.0001566817],"about_ca_topic_score_codex":0.019459326,"about_ca_topic_score_gemma":0.034860857,"teacher_disagreement_score":0.039241903,"about_ca_system_score_codex":0.002639262,"about_ca_system_score_gemma":0.00443022,"threshold_uncertainty_score":0.13127726},"labels":[],"label_agreement":null},{"id":"W7039326777","doi":"","title":"Le modÃ¨le antarctique","year":2012,"lang":"fr","type":"other","venue":"Library and Archives Canada (Government of Canada)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Context (archaeology)","score_opus":0.005637717421746731,"score_gpt":0.14927996539349433,"score_spread":0.1436422479717476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7039326777","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023636403,0.0030238885,0.0033573634,0.0035741844,0.0017857956,0.00004923515,0.00069626916,0.00017773261,0.96369916],"genre_scores_gemma":[0.25889933,0.006289003,0.005851656,0.0012200547,0.00068886334,0.00008861606,0.0015409314,0.0004096424,0.72501194],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999094,0.00013049268,0.00003617774,0.0001981493,0.00040805555,0.00013317002],"domain_scores_gemma":[0.9996878,0.00006029667,0.00003961327,0.00005333803,0.00010925888,0.00004974525],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051136443,0.0005529272,0.00033676717,0.000763576,0.0034122309,0.0056549376,0.00054950826,0.0010071563,0.051718544],"category_scores_gemma":[0.0011772083,0.00022673748,0.00040350886,0.0016092246,0.0023398006,0.0020213365,0.002236416,0.0024253111,0.011727821],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016116234,0.000050233768,0.011501033,0.0005185501,0.000049317157,0.0019030664,0.016341625,0.0015839277,0.003923303,0.6834172,0.0748942,0.20565635],"study_design_scores_gemma":[0.0000043761397,0.000011074099,0.0027750502,0.00006675372,0.0000030515146,0.00023950585,0.0012524162,0.00010647242,0.00025335103,0.0037674673,0.99151284,0.000007696268],"about_ca_topic_score_codex":0.038770255,"about_ca_topic_score_gemma":0.05708845,"teacher_disagreement_score":0.051718544,"about_ca_system_score_codex":0.0048312712,"about_ca_system_score_gemma":0.0044193137,"threshold_uncertainty_score":0.17301577},"labels":[],"label_agreement":null},{"id":"W7039669055","doi":"","title":"Mechanická a chemická regulace pcháče rolního","year":2007,"lang":"en","type":"dissertation","venue":"Digital Repository (National Repository of Grey Literature)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Thistle; Chemical control; Arable land; Vegetation (pathology); Perennial plant; Weed control","score_opus":0.008220208854451792,"score_gpt":0.2602790487211613,"score_spread":0.2520588398667095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7039669055","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02780437,0.07622256,0.018055683,0.017192433,0.003546653,0.00065911264,0.005536463,0.0025629466,0.84841985],"genre_scores_gemma":[0.07515423,0.038000997,0.011831333,0.0013515807,0.0003127579,0.00018108093,0.0018888601,0.00045889302,0.8708202],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99945253,0.000022018587,0.000013612327,0.000107629756,0.00030326258,0.00010104036],"domain_scores_gemma":[0.9996816,0.000022196728,0.00001855542,0.000026705256,0.000151801,0.00009915321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005777745,0.0007090904,0.0006343572,0.0009305537,0.0015479589,0.0026922172,0.0005843608,0.00071153574,0.13601772],"category_scores_gemma":[0.0005704565,0.00035500852,0.0003046676,0.0008578239,0.000882191,0.00094462495,0.0013647368,0.0018319198,0.039427005],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005605521,0.00044275462,0.0012711526,0.0025696668,0.000049554747,0.00033420423,0.0012430517,0.0012688345,0.06588063,0.05168854,0.24145475,0.6332362],"study_design_scores_gemma":[0.000027457958,0.00006861724,0.0012352896,0.00016402613,0.0000090796375,0.00010059214,0.00014470729,0.00018633202,0.0065877205,0.0010489811,0.9904141,0.0000130382705],"about_ca_topic_score_codex":0.051548004,"about_ca_topic_score_gemma":0.08872955,"teacher_disagreement_score":0.13601772,"about_ca_system_score_codex":0.0039081057,"about_ca_system_score_gemma":0.0056878617,"threshold_uncertainty_score":0.4550246},"labels":[],"label_agreement":null},{"id":"W7042304904","doi":"","title":"Petition for Truing Up of transmission tariff under “System Strengthening Scheme in Roorkee in Northern Region” – EQ","year":2025,"lang":"en","type":"other","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Scheme (mathematics); Tariff; Transmission (telecommunications); Quarter (Canadian coin)","score_opus":0.02588655558605262,"score_gpt":0.26128980982712,"score_spread":0.2354032542410674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7042304904","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08987094,0.0008509327,0.0067180763,0.086944826,0.007121773,0.0014183491,0.0069235223,0.002315187,0.79783636],"genre_scores_gemma":[0.05112274,0.000107510125,0.001793564,0.0181376,0.00025669832,0.00015962964,0.0010913518,0.00020568383,0.92712516],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99845505,0.00017990211,0.000054694272,0.00019320419,0.00053996115,0.0005771049],"domain_scores_gemma":[0.99801147,0.00031980613,0.0000879565,0.00012517173,0.00088033173,0.0005753272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019718064,0.00023290102,0.00023515352,0.00057204417,0.0028099634,0.0023073645,0.0010954984,0.005490847,0.062476255],"category_scores_gemma":[0.003995097,0.00033817638,0.00050156284,0.00027264236,0.00066534174,0.0008170098,0.001309065,0.00444083,0.010631671],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049344316,0.0003169598,0.009372617,0.00015583448,0.00005015513,0.0019645453,0.00072358066,0.00059734815,0.009245198,0.043521866,0.89257,0.04098848],"study_design_scores_gemma":[0.00011740554,0.00015798959,0.019580688,0.000067090274,0.000023621058,0.0003382894,0.0005066372,0.0008322732,0.0020530967,0.0015972172,0.9746816,0.000044190587],"about_ca_topic_score_codex":0.145584,"about_ca_topic_score_gemma":0.2239305,"teacher_disagreement_score":0.145584,"about_ca_system_score_codex":0.0033514474,"about_ca_system_score_gemma":0.01041219,"threshold_uncertainty_score":0.28947318},"labels":[],"label_agreement":null},{"id":"W7084066819","doi":"10.1007/978-3-032-00006-4_2","title":"Trichomorphology","year":2025,"lang":"en","type":"book-chapter","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Museum of Nature","funders":"","keywords":"Focus (optics); Diversity (politics); Taxonomy (biology); Line drawings; Biological evolution","score_opus":0.02019131002627012,"score_gpt":0.24671869105733785,"score_spread":0.22652738103106773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7084066819","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017473185,0.001787286,0.015419277,0.0009282196,0.0005300927,0.000025668125,0.00013658452,0.00013132126,0.97929424],"genre_scores_gemma":[0.050805047,0.002940509,0.008166142,0.00062451506,0.0005098106,0.00006600992,0.0003400131,0.00037233072,0.9361757],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998235,0.000031261585,0.000006325746,0.000048809445,0.00006485346,0.000025334493],"domain_scores_gemma":[0.9998306,0.00004070591,0.00000833186,0.000050457635,0.000050103277,0.000019787629],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00015855485,0.00056234933,0.00044566873,0.0014472757,0.0018317533,0.0023809446,0.000889145,0.00074875157,0.06861769],"category_scores_gemma":[0.0005304984,0.00030044987,0.00039399456,0.0010129431,0.0034624115,0.0030079558,0.0013976437,0.0025069593,0.021006405],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013243122,0.00001153352,0.000080033,0.00006519034,0.0000021414571,0.00006268021,0.00038857202,0.000101770966,0.00074258924,0.9000465,0.04685132,0.051634587],"study_design_scores_gemma":[0.000003661167,0.000012074981,0.00021719186,0.0000427573,0.0000037862933,0.0004212439,0.00030653158,0.0002394019,0.0004701108,0.27101597,0.7272611,0.0000061950677],"about_ca_topic_score_codex":0.0016831957,"about_ca_topic_score_gemma":0.0031775166,"teacher_disagreement_score":0.06861769,"about_ca_system_score_codex":0.0015888338,"about_ca_system_score_gemma":0.0005752622,"threshold_uncertainty_score":0.22954905},"labels":[],"label_agreement":null},{"id":"W7091358849","doi":"10.1109/access.2025.3622024","title":"Robust Real-Time Arabic Speech Recognition for AAVs in Adverse Acoustic Conditions Using Lightweight CNNs","year":2025,"lang":"en","type":"article","venue":"IEEE Access","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Convolutional neural network; Latency (audio); Inference; Noise (video); Robustness (evolution); Pattern recognition (psychology); Arabic; Ranging; Speech enhancement","score_opus":0.07700867099233005,"score_gpt":0.34108491656649564,"score_spread":0.2640762455741656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7091358849","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16911064,0.0021594933,0.80353653,0.0004310874,0.0007321433,0.00018697682,0.0009807856,0.014070214,0.008792157],"genre_scores_gemma":[0.84117556,0.0007675975,0.1421975,0.00035838655,0.00014537368,0.00016780668,0.0020806855,0.0002753385,0.012831679],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99962807,0.000034919918,0.00001787833,0.00014111525,0.000109236935,0.000068687106],"domain_scores_gemma":[0.99972385,0.000068568545,0.000028817587,0.00004369602,0.00011683808,0.00001824581],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035687588,0.0012444716,0.0005326203,0.00041876355,0.0002600793,0.0006187812,0.000879119,0.00050125027,0.0028174007],"category_scores_gemma":[0.0010104968,0.00030601953,0.0005070265,0.00021433138,0.00028197706,0.00072001276,0.000783606,0.00092076487,0.0023655263],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060687365,0.00015317209,0.0019714115,0.00020769292,0.00010844346,0.00031582286,0.00012831077,0.07808997,0.16105044,0.0011016617,0.006318446,0.7499478],"study_design_scores_gemma":[0.000017360275,0.00013651139,0.002291292,0.000030590225,0.000059917144,0.00016672963,0.00007295432,0.93125105,0.06074304,0.00083009043,0.004370333,0.000030092366],"about_ca_topic_score_codex":0.008548735,"about_ca_topic_score_gemma":0.013850675,"teacher_disagreement_score":0.008548735,"about_ca_system_score_codex":0.00045919736,"about_ca_system_score_gemma":0.00066744833,"threshold_uncertainty_score":0.016997933},"labels":[],"label_agreement":null},{"id":"W7092191125","doi":"10.1109/ms.2025.3621625","title":"What Inputs Drive Effective Large Language Model-Based Unit Test Generation?","year":2025,"lang":"","type":"article","venue":"IEEE Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Unit testing; Correctness; Software quality; Code coverage; Code (set theory); Test (biology); Test Management Approach; Test case","score_opus":0.01701364261331757,"score_gpt":0.29850441364940394,"score_spread":0.2814907710360864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7092191125","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6786712,0.0008875203,0.3036198,0.0015989251,0.00006992648,0.00039418603,0.0005101923,0.005830725,0.008417576],"genre_scores_gemma":[0.9405611,0.00016283954,0.057865642,0.00019609273,0.00002086812,0.00013205747,0.0004342835,0.00034696487,0.0002801587],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99400663,0.0032215347,0.00030991583,0.00066942966,0.0014073414,0.0003851378],"domain_scores_gemma":[0.93310267,0.052630324,0.00430342,0.004259464,0.0048099477,0.0008941324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056726667,0.0011466936,0.0009301038,0.0013068302,0.00056003645,0.003236014,0.0017212884,0.0013361215,0.0026067584],"category_scores_gemma":[0.090160705,0.00075713056,0.0005885959,0.00101147,0.0011585547,0.004764884,0.0019456234,0.0015294563,0.00096210296],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016796775,0.00214076,0.12612861,0.0016400537,0.00022139796,0.0012894665,0.0022145489,0.22878164,0.1263768,0.018434705,0.006346722,0.48474565],"study_design_scores_gemma":[0.00011519927,0.0008430194,0.021762494,0.00019566888,0.00014748804,0.00031513424,0.00071415777,0.8972227,0.05724274,0.018807812,0.002571489,0.00006208777],"about_ca_topic_score_codex":0.0019163048,"about_ca_topic_score_gemma":0.002774842,"teacher_disagreement_score":0.0056726667,"about_ca_system_score_codex":0.0014025297,"about_ca_system_score_gemma":0.0015339163,"threshold_uncertainty_score":0.03000027},"labels":[],"label_agreement":null},{"id":"W7097569464","doi":"","title":"M.: Formalizing a structured natural language requirements specification notation","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Notation; Formal specification; Readability; B-Method; Specification language; Programming language specification; Software requirements specification; Formal methods","score_opus":0.03928953644046304,"score_gpt":0.2763948160140021,"score_spread":0.23710527957353908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7097569464","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031112337,0.00007373161,0.9895103,0.00050583086,0.00006755973,0.00029386434,0.0003617238,0.0015164835,0.004559175],"genre_scores_gemma":[0.029907735,0.00012580618,0.9660931,0.00027229133,0.00003234781,0.00054693077,0.0005842756,0.00018047959,0.002257015],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9952242,0.0025106396,0.0006949105,0.0005288935,0.0008470697,0.00019437006],"domain_scores_gemma":[0.992448,0.0044712955,0.0010374031,0.00096428354,0.0008881235,0.00019078553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009069356,0.001184786,0.0004482109,0.0011852336,0.0009272727,0.002688947,0.0015646301,0.0015834603,0.0039159246],"category_scores_gemma":[0.012867006,0.00082853495,0.0015808202,0.0006973855,0.0020648548,0.0036451172,0.002277284,0.0022568314,0.0017621642],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001615974,0.00011854941,0.00075060996,0.00057063974,0.000034625624,0.00080892147,0.0017201871,0.024070516,0.014722594,0.868818,0.01166752,0.07655623],"study_design_scores_gemma":[0.0002975807,0.00045093943,0.0005134264,0.00067991577,0.000108229484,0.001492351,0.00062537397,0.30431435,0.05108004,0.29470837,0.34555793,0.0001715464],"about_ca_topic_score_codex":0.0021243426,"about_ca_topic_score_gemma":0.002806078,"teacher_disagreement_score":0.009069356,"about_ca_system_score_codex":0.0011703166,"about_ca_system_score_gemma":0.0030837478,"threshold_uncertainty_score":0.047963917},"labels":[],"label_agreement":null},{"id":"W7099317437","doi":"","title":"Contents","year":2002,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Relation (database); Identification (biology); Class (philosophy); Set (abstract data type)","score_opus":0.05935730714539072,"score_gpt":0.24303471369861357,"score_spread":0.18367740655322284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7099317437","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010958178,0.0017589632,0.0020916183,0.0069854986,0.005575016,0.00015882734,0.0070782965,0.0011250158,0.974131],"genre_scores_gemma":[0.0084116785,0.0023911747,0.0010186561,0.001531332,0.0015677407,0.00008657874,0.006228951,0.00059032755,0.97817343],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993937,0.0000802282,0.00002085273,0.00008654665,0.00036192546,0.000056690245],"domain_scores_gemma":[0.9975217,0.0003096557,0.000115232826,0.0002580789,0.0014384001,0.00035698284],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00051152287,0.0004445208,0.00049465464,0.0017290101,0.0014089053,0.0043373234,0.0007109173,0.0008333184,0.6855993],"category_scores_gemma":[0.004257877,0.00018287315,0.00024500667,0.0016400159,0.0004352659,0.0021528013,0.0016222694,0.0009935901,0.5734339],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030291849,0.000021111091,0.00029375817,0.00008365164,0.0000020494351,0.000025254622,0.00011167245,0.00007679404,0.0005034172,0.008763286,0.9064719,0.08361686],"study_design_scores_gemma":[0.0000014299372,0.000005365628,0.00023199286,0.000048475777,0.0000015785226,0.000023224642,0.00008336114,0.00003853443,0.0001547854,0.0010755658,0.9983335,0.0000022284826],"about_ca_topic_score_codex":0.0040530767,"about_ca_topic_score_gemma":0.0032088645,"teacher_disagreement_score":0.31440067,"about_ca_system_score_codex":0.0022318296,"about_ca_system_score_gemma":0.0016363272,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W7099896649","doi":"","title":"Human Resources and","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Human capital; Workforce; Population ageing; Human resources; Population; Ageing society; Labour supply; Capital (architecture)","score_opus":0.012969221405160066,"score_gpt":0.24422599092828162,"score_spread":0.23125676952312155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7099896649","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008772271,0.0052505303,0.0021306311,0.02349283,0.00072928105,0.00006346916,0.000888861,0.0000702395,0.9586019],"genre_scores_gemma":[0.36421812,0.009130489,0.0026479235,0.007952514,0.0005609475,0.00017834196,0.0012068519,0.00007598191,0.61402893],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99842983,0.0004375859,0.000055564302,0.00027357013,0.0003913042,0.00041204967],"domain_scores_gemma":[0.9987741,0.00020091132,0.00013976987,0.0001718351,0.00030557494,0.00040777982],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0011302157,0.0003661336,0.00031069314,0.0011092076,0.002869196,0.00575443,0.00077525433,0.0013420896,0.12243029],"category_scores_gemma":[0.0035260394,0.00009684851,0.00027138894,0.0019741773,0.0028108705,0.0025108256,0.0033561783,0.0011355646,0.014782553],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000372315,0.000067954985,0.0054095164,0.0001399312,0.00002191275,0.00024475966,0.0027803136,0.0003468141,0.00015415084,0.7318469,0.12106321,0.13788737],"study_design_scores_gemma":[0.0000065271383,0.000024928333,0.0065122293,0.00028142033,0.0000060842112,0.00027355648,0.0032510771,0.00013415144,0.0000836783,0.054121528,0.9352912,0.000013551646],"about_ca_topic_score_codex":0.022136537,"about_ca_topic_score_gemma":0.027472964,"teacher_disagreement_score":0.87756974,"about_ca_system_score_codex":0.0038157117,"about_ca_system_score_gemma":0.0055720173,"threshold_uncertainty_score":0.40957016},"labels":[],"label_agreement":null},{"id":"W7101429739","doi":"10.2139/ssrn.5664811","title":"Were you on Facebook 10 years ago? You may be able to claim part of this $50 million payout","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Payment; Compensation (psychology); Financial compensation; Quarter (Canadian coin)","score_opus":0.023074247127207587,"score_gpt":0.2844251055892946,"score_spread":0.26135085846208705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7101429739","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02755453,0.004722223,0.004217287,0.11205214,0.04905202,0.00027549037,0.004661633,0.0050746044,0.7923901],"genre_scores_gemma":[0.025139287,0.0006446311,0.00042618753,0.0043808417,0.0016089234,0.00007079708,0.00077374326,0.00035859368,0.96659696],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992004,0.00009811269,0.000024190518,0.000085700936,0.00035906618,0.00023264936],"domain_scores_gemma":[0.995865,0.00046842935,0.00019058678,0.0005089439,0.00075643824,0.0022105945],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0009442141,0.00055626885,0.0008079015,0.0011719872,0.0033279432,0.0046152202,0.00058993313,0.0035520617,0.4926418],"category_scores_gemma":[0.0075696427,0.0003625783,0.00048170576,0.00071474013,0.0007828805,0.0034032417,0.0031844417,0.0027822421,0.3175029],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011871433,0.000102708414,0.0011857748,0.000025216295,0.000007884226,0.0001580225,0.00008236773,0.000043521177,0.00055356836,0.005416841,0.9383156,0.05398977],"study_design_scores_gemma":[0.00002930584,0.00007692683,0.0032035003,0.00006130568,0.000011377903,0.00028012795,0.00037111773,0.00032179404,0.0003623245,0.0026122006,0.9926443,0.000025681642],"about_ca_topic_score_codex":0.0026595881,"about_ca_topic_score_gemma":0.0059452527,"teacher_disagreement_score":0.4926418,"about_ca_system_score_codex":0.000789884,"about_ca_system_score_gemma":0.00090973836,"threshold_uncertainty_score":0.7236849},"labels":[],"label_agreement":null},{"id":"W7105692598","doi":"10.1109/ijcnn64981.2025.11227916","title":"Model Cascading for Code: A Cascaded Black-Box Multi-Model Framework for Cost-Efficient Code Completion with Self-Testing","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University","keywords":"Code (set theory); Heuristic; Server; Sensitivity (control systems); Test case; Source code; Computational complexity theory; Component (thermodynamics)","score_opus":0.1273860547350612,"score_gpt":0.3602348113149539,"score_spread":0.23284875657989268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7105692598","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016693668,0.00021691735,0.9750574,0.00024750823,0.000037757007,0.00017641917,0.00007637365,0.006445697,0.0010482378],"genre_scores_gemma":[0.42587098,0.00014462964,0.5692305,0.0003034979,0.000044761287,0.00046382064,0.00040394053,0.0016414329,0.0018963836],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967056,0.0012792429,0.00014561292,0.0006057677,0.0008860875,0.0003776195],"domain_scores_gemma":[0.9882676,0.0072867316,0.0007494765,0.0021091676,0.0010731649,0.0005138947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004580134,0.0021675,0.0016529771,0.0014333117,0.00090297393,0.0018408097,0.0054796175,0.002098745,0.0060881716],"category_scores_gemma":[0.019346343,0.0014681449,0.0023891644,0.000759605,0.0022372846,0.0033035323,0.0036832355,0.0035626716,0.0013228693],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021273633,0.00020753003,0.0024874422,0.00021330596,0.000090071015,0.00029633357,0.00021868825,0.87634915,0.00490596,0.018207563,0.0025349015,0.094276235],"study_design_scores_gemma":[0.0000123057025,0.0000265154,0.000036142093,0.000005913819,0.0000073446163,0.00001830403,0.000007214555,0.9926011,0.0007816168,0.006229032,0.0002692614,0.0000051804514],"about_ca_topic_score_codex":0.006460168,"about_ca_topic_score_gemma":0.0077502807,"teacher_disagreement_score":0.006460168,"about_ca_system_score_codex":0.0017787014,"about_ca_system_score_gemma":0.0029127172,"threshold_uncertainty_score":0.024222374},"labels":[],"label_agreement":null},{"id":"W7106150544","doi":"10.48550/arxiv.2511.14432","title":"Mutation Testing for Industrial Robotic Systems","year":2025,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Chicoutimi","funders":"","keywords":"Mutation; Software; Reliability (semiconductor); Process (computing); Robot; Industrial robot; Quality (philosophy); Mutation testing","score_opus":0.24057651389724458,"score_gpt":0.23107414152037709,"score_spread":0.00950237237686749,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7106150544","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17837,0.0008219724,0.81465083,0.0005226202,0.00005499117,0.00012287786,0.00009412804,0.0029582782,0.0024042523],"genre_scores_gemma":[0.803994,0.00036283175,0.19398524,0.00018039437,0.000028742343,0.0001259966,0.000156233,0.0002393265,0.0009271699],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963971,0.0015133999,0.00018009136,0.00040515995,0.0013345763,0.00016968591],"domain_scores_gemma":[0.99011606,0.007481307,0.00083919807,0.0006932928,0.0007189435,0.00015124556],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020695748,0.0008069901,0.00052070286,0.001319093,0.00043168932,0.0006689883,0.001257559,0.00091705436,0.0007694583],"category_scores_gemma":[0.012991275,0.00023834892,0.0006135014,0.0008151539,0.0020407604,0.0010368337,0.0008587906,0.0010034733,0.00012522379],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038492787,0.0003447873,0.011595261,0.0005589185,0.00011348006,0.0016977208,0.0005710211,0.53351134,0.08965633,0.070450746,0.0019246761,0.2891907],"study_design_scores_gemma":[0.00007181184,0.00030604046,0.0016011108,0.00007945822,0.0000439532,0.0007074467,0.000072129835,0.8809345,0.044835817,0.068030216,0.0032783302,0.00003908906],"about_ca_topic_score_codex":0.002070746,"about_ca_topic_score_gemma":0.0013181581,"teacher_disagreement_score":0.002070746,"about_ca_system_score_codex":0.00093676825,"about_ca_system_score_gemma":0.0008932682,"threshold_uncertainty_score":0.010945082},"labels":[],"label_agreement":null},{"id":"W7106650730","doi":"10.32913/mic-ict-research.v2025.n1.1277","title":"A Rich High-Order Mutation Testing Dataset for Software Fortification","year":2024,"lang":"","type":"article","venue":"Research and Development on Information and Communication Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Mutation; Mutation testing; Random forest; Software; Software testing; Test (biology)","score_opus":0.11383902518043638,"score_gpt":0.3782881379797234,"score_spread":0.264449112799287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7106650730","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.62484103,0.0027889267,0.029882267,0.0018333553,0.00036963186,0.0006545252,0.32272443,0.0074311,0.009474747],"genre_scores_gemma":[0.40265486,0.0006189396,0.052785054,0.00061913766,0.000112168396,0.0007274001,0.53874147,0.00047635165,0.0032646165],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983961,0.0002464854,0.00016039831,0.0003787552,0.0006863743,0.00013182874],"domain_scores_gemma":[0.9945273,0.0022028964,0.0006684033,0.0009846895,0.0012706215,0.0003461528],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012168132,0.0010479565,0.0005243696,0.0025103528,0.0008091828,0.00075972266,0.0021993984,0.001984756,0.0013676476],"category_scores_gemma":[0.006269273,0.00027141895,0.00085572206,0.0024281843,0.0005965206,0.000743907,0.0009736361,0.0016181082,0.0009250327],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014609577,0.003375338,0.16170947,0.0033402145,0.0006572278,0.005241799,0.0010395676,0.12735216,0.031163217,0.0072369673,0.4136482,0.24377492],"study_design_scores_gemma":[0.00082712236,0.0013311418,0.32599473,0.0005563315,0.00028064582,0.00500951,0.0010337768,0.26411536,0.054036967,0.013818815,0.3326749,0.00032074045],"about_ca_topic_score_codex":0.008647153,"about_ca_topic_score_gemma":0.020742755,"teacher_disagreement_score":0.008647153,"about_ca_system_score_codex":0.00093148113,"about_ca_system_score_gemma":0.0013288821,"threshold_uncertainty_score":0.017193615},"labels":[],"label_agreement":null},{"id":"W7117154141","doi":"10.1145/3756681.3756991","title":"PRIMG : Efficient LLM-driven Test Generation Using Mutant Prioritization","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Prioritization; Test case; Process (computing); Mutation; Code coverage; Code (set theory); Test (biology); Software","score_opus":0.03586475559047253,"score_gpt":0.2990342536693706,"score_spread":0.26316949807889806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117154141","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014839774,0.00013126947,0.96167487,0.00018229445,0.000028553608,0.00024236241,0.00018692059,0.021740394,0.00097362924],"genre_scores_gemma":[0.31413,0.000109879635,0.6813448,0.00028909536,0.000024124307,0.00039265442,0.00097216095,0.0014605441,0.0012767893],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99719244,0.0009782242,0.00014596594,0.00045318232,0.0009960884,0.00023412587],"domain_scores_gemma":[0.9944634,0.003191144,0.000517041,0.0009849684,0.00067701703,0.00016642152],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002920238,0.0014500867,0.0008297691,0.0016188216,0.00035056658,0.0009125098,0.0024863984,0.0010114287,0.0027770188],"category_scores_gemma":[0.012421538,0.00059357577,0.0009810593,0.0006427039,0.0010819399,0.0014059666,0.0020928422,0.0014860746,0.0009954078],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000402559,0.0004246017,0.008781786,0.00041620852,0.00011970343,0.0007115957,0.000316272,0.27813962,0.048939094,0.01822348,0.011589313,0.6319357],"study_design_scores_gemma":[0.000056819375,0.00009049169,0.00041314744,0.000019664865,0.000021534284,0.00014287877,0.00001865127,0.9699533,0.01742909,0.009479126,0.0023568254,0.000018486773],"about_ca_topic_score_codex":0.003519351,"about_ca_topic_score_gemma":0.0044456352,"teacher_disagreement_score":0.003519351,"about_ca_system_score_codex":0.001076692,"about_ca_system_score_gemma":0.0022682922,"threshold_uncertainty_score":0.0154438615},"labels":[],"label_agreement":null},{"id":"W7117322255","doi":"10.1016/j.jss.2025.112759","title":"On software testing reference ontologies","year":2025,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Ontology; Process ontology; IDEF5; Formality; Upper ontology; Domain (mathematical analysis); Ontology components; Ontology-based data integration","score_opus":0.04469546520606831,"score_gpt":0.2869665398924838,"score_spread":0.24227107468641546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117322255","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020750636,0.0023068835,0.92778593,0.0057821306,0.0003654084,0.00014695678,0.00041578332,0.0013914551,0.041054823],"genre_scores_gemma":[0.47913867,0.004264561,0.4874615,0.0023984173,0.0006508197,0.00033132551,0.003033068,0.0017723663,0.020949231],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98443514,0.0053721806,0.0014706256,0.0018551998,0.0059086075,0.0009581551],"domain_scores_gemma":[0.9361924,0.030510198,0.0019933158,0.017759398,0.01264337,0.0009012598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012030764,0.0009425127,0.0017806286,0.008824454,0.003679173,0.005846318,0.004056445,0.0038753818,0.0074472167],"category_scores_gemma":[0.07170849,0.0010485667,0.001627804,0.011570462,0.005561886,0.031630684,0.008116286,0.0047442205,0.0022008743],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000050669376,0.000056381374,0.0007501275,0.00010996142,0.000017877537,0.0001448131,0.0007629706,0.0045043635,0.00059613195,0.9034621,0.005818993,0.08372552],"study_design_scores_gemma":[0.000013714519,0.00003171519,0.0004171705,0.00030108046,0.000045687997,0.00020797952,0.00044282345,0.024815096,0.0016606633,0.9292396,0.042791747,0.000032678156],"about_ca_topic_score_codex":0.01339817,"about_ca_topic_score_gemma":0.010690585,"teacher_disagreement_score":0.01339817,"about_ca_system_score_codex":0.002919751,"about_ca_system_score_gemma":0.0032871298,"threshold_uncertainty_score":0.063625515},"labels":[],"label_agreement":null},{"id":"W7124996121","doi":"10.1109/aiware69974.2025.00012","title":"Turning Manual Tasks Into Actions: Assessing the Effectiveness of Gemini-Generated Selenium Tests","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Executable; HTML; Software; Test (biology); Hypertext; User interface","score_opus":0.02708333145676899,"score_gpt":0.3548974381655703,"score_spread":0.3278141067088013,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7124996121","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9760572,0.0003972346,0.015888648,0.00009796173,0.0000461076,0.00025696697,0.00046097688,0.004017249,0.0027777886],"genre_scores_gemma":[0.9539682,0.00015188611,0.040954396,0.00011649289,0.00001824055,0.00022585278,0.002311771,0.00084091094,0.0014122559],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9911788,0.003474722,0.001089437,0.0012115785,0.0026493296,0.00039612257],"domain_scores_gemma":[0.8794248,0.09667136,0.0075944914,0.008544275,0.006432122,0.0013330146],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0080935005,0.0013556528,0.0006403725,0.002197915,0.00032507084,0.0012851775,0.0014529763,0.00096627633,0.0013265788],"category_scores_gemma":[0.089702554,0.00046360964,0.00048001413,0.0009295448,0.00082994305,0.0011521906,0.001322028,0.0008166509,0.0008152339],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009363849,0.0050608,0.13982287,0.002194406,0.0005894634,0.0008915826,0.0051351874,0.12960972,0.08985608,0.0018903401,0.0070819072,0.60850376],"study_design_scores_gemma":[0.00089836377,0.01769195,0.18495184,0.00039622094,0.00056644855,0.0011274948,0.0026174884,0.53332037,0.23918964,0.0022753414,0.01656953,0.00039531698],"about_ca_topic_score_codex":0.0024713592,"about_ca_topic_score_gemma":0.0023684448,"teacher_disagreement_score":0.0080935005,"about_ca_system_score_codex":0.00069812656,"about_ca_system_score_gemma":0.00061823445,"threshold_uncertainty_score":0.04280305},"labels":[],"label_agreement":null},{"id":"W7125608420","doi":"10.1109/cascon66301.2025.00085","title":"Optimizing Test Case Reduction in CI/CD Pipelines Using Integer Linear Programming","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); Ontario Tech University","funders":"","keywords":"Test suite; Test case; Reduction (mathematics); System under test; Pipeline transport; Test (biology); Integer programming; Test Management Approach; Software deployment; Software","score_opus":0.04453385423626503,"score_gpt":0.33276822099642644,"score_spread":0.2882343667601614,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125608420","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11822149,0.00044835766,0.87474346,0.00046486538,0.000031312902,0.00023410683,0.00018628311,0.002042187,0.003627903],"genre_scores_gemma":[0.6294275,0.00015975583,0.36726823,0.00023002012,0.000027168555,0.00032836478,0.00053850847,0.00035030555,0.0016702049],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985233,0.0005982694,0.00005732078,0.0001847377,0.00040117177,0.00023525738],"domain_scores_gemma":[0.9940655,0.0047836266,0.00044674258,0.00018325573,0.00039727648,0.00012365397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019433155,0.0018737743,0.0010637576,0.0013403562,0.0003588772,0.0008964882,0.0012282998,0.00072339387,0.0018881902],"category_scores_gemma":[0.007432167,0.00080589467,0.00085323513,0.0010485796,0.0010420082,0.000797317,0.0008421471,0.001735769,0.00026782835],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008000774,0.00008837622,0.0007942543,0.000056591318,0.000018887686,0.000056574914,0.000020425692,0.9623719,0.00194575,0.0009750567,0.0004385308,0.033153567],"study_design_scores_gemma":[0.0000092950195,0.000034593402,0.000099428195,0.0000034092711,0.000005205175,0.000010707638,0.00000767138,0.9979424,0.00085801363,0.00091842905,0.00010863223,0.0000022254076],"about_ca_topic_score_codex":0.007939548,"about_ca_topic_score_gemma":0.007523961,"teacher_disagreement_score":0.007939548,"about_ca_system_score_codex":0.0013079814,"about_ca_system_score_gemma":0.0021391425,"threshold_uncertainty_score":0.015786648},"labels":[],"label_agreement":null},{"id":"W7125807424","doi":"10.17504/protocols.io.6qpvryedogmk/v1","title":"Generation of Mutagenic Libraries using a POPCode Method v1","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Subcloning; Amplicon; Gene; DNA; Protocol (science); Ensembl","score_opus":0.13088762140403756,"score_gpt":0.3680847536034365,"score_spread":0.23719713219939895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125807424","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12143498,0.00079135556,0.83819664,0.00022840388,0.0002485494,0.0053394223,0.008581851,0.016195206,0.00898351],"genre_scores_gemma":[0.24512559,0.0025258833,0.66182095,0.0005410969,0.00005855908,0.0070742727,0.037466932,0.0061918115,0.03919494],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99885714,0.0001859211,0.00013321108,0.00025179243,0.00042841266,0.0001434878],"domain_scores_gemma":[0.9993262,0.00018337488,0.000095977106,0.00021122891,0.00011451627,0.00006866741],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091814325,0.0012739762,0.0010761387,0.0015117372,0.0007249695,0.0008835345,0.0012019646,0.00084772124,0.0069361944],"category_scores_gemma":[0.0013155488,0.0010311238,0.00081255514,0.0008880312,0.00041781794,0.00042813295,0.0011678943,0.00186417,0.007527915],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012138454,0.00008347476,0.0002268995,0.00016767066,0.000024846031,0.00015916974,0.00006883769,0.00037694542,0.98186857,0.00077872956,0.0010533745,0.015070103],"study_design_scores_gemma":[0.000031244042,0.00022021087,0.0006823887,0.00002154611,0.000038142414,0.00044547374,0.000017023094,0.001490646,0.9744125,0.00020770659,0.022394681,0.000038493148],"about_ca_topic_score_codex":0.00059956644,"about_ca_topic_score_gemma":0.001339378,"teacher_disagreement_score":0.0069361944,"about_ca_system_score_codex":0.0002964496,"about_ca_system_score_gemma":0.000776223,"threshold_uncertainty_score":0.02320385},"labels":[],"label_agreement":null},{"id":"W7125894626","doi":"10.1109/ase63991.2025.00308","title":"MobileUPReg: Identifying User-Perceived Performance Regressions in Mobile OS Versions","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Concordia University","funders":"","keywords":"Wilcoxon signed-rank test; Regression; Mobile device; Focus (optics); Regression analysis; Regression testing; Baseline (sea); Linear regression","score_opus":0.027014622866123644,"score_gpt":0.3113889806829244,"score_spread":0.28437435781680076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125894626","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7325167,0.002814326,0.16053282,0.0002064071,0.00020323826,0.0004892764,0.0038652094,0.095421016,0.003950996],"genre_scores_gemma":[0.9388454,0.00030820415,0.05475215,0.00019937698,0.00004674313,0.00018434116,0.0023361207,0.0013992976,0.0019283557],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99818987,0.00024046106,0.000119207034,0.00056348875,0.00072808674,0.0001588496],"domain_scores_gemma":[0.99318814,0.0028167928,0.0013908369,0.0009862453,0.0013788831,0.00023917724],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014307676,0.0022542549,0.00085233356,0.0025276742,0.00030954008,0.0011230353,0.0010885907,0.0007898475,0.0010969416],"category_scores_gemma":[0.012445825,0.0004149337,0.000441846,0.0008013169,0.0002488092,0.0013155413,0.000997475,0.00069982087,0.0012923927],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027187187,0.0007144858,0.19245483,0.0012674592,0.0004556459,0.0007102375,0.001332619,0.0066422783,0.11814438,0.0011026888,0.02694392,0.64751273],"study_design_scores_gemma":[0.00018561854,0.003194001,0.3968208,0.0003064842,0.00043501283,0.0020438584,0.00077889324,0.38825813,0.18153937,0.002464648,0.023608262,0.00036491652],"about_ca_topic_score_codex":0.0030671763,"about_ca_topic_score_gemma":0.0045250216,"teacher_disagreement_score":0.0030671763,"about_ca_system_score_codex":0.00033732018,"about_ca_system_score_gemma":0.0004063619,"threshold_uncertainty_score":0.0075666904},"labels":[],"label_agreement":null},{"id":"W7125931302","doi":"10.1109/ase63991.2025.00365","title":"PrioTestCI: Efficient Test Case Prioritization in GitHub Workflows for CI Optimization","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Workflow; Test suite; Test case; Test (biology); Prioritization; Software; Regression testing; Code (set theory); Reduction (mathematics)","score_opus":0.01531142197187153,"score_gpt":0.2894594539366552,"score_spread":0.27414803196478366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125931302","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025189165,0.0006244028,0.6295909,0.0006549172,0.0002971704,0.00079694757,0.0011352143,0.3341259,0.0075853216],"genre_scores_gemma":[0.20637073,0.00034894847,0.7281057,0.0007070268,0.000105842395,0.0011571605,0.0058512134,0.052181713,0.0051715523],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99076957,0.0029503226,0.0007357246,0.0015357289,0.002971506,0.0010372085],"domain_scores_gemma":[0.9809188,0.0070492737,0.0015153283,0.0064623062,0.003112407,0.00094172003],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068474994,0.0038819737,0.0012011171,0.0032502196,0.0013771515,0.0032401264,0.0054806024,0.0013187639,0.009694962],"category_scores_gemma":[0.031334914,0.0019426515,0.0020733066,0.0018597258,0.002402181,0.003733378,0.0054853605,0.004112405,0.0060204053],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017679428,0.0007649544,0.01575286,0.001305423,0.00033751625,0.0013281155,0.002091651,0.047440194,0.06362426,0.020218201,0.1366456,0.7087233],"study_design_scores_gemma":[0.00082623574,0.00085055694,0.009911482,0.00056573487,0.00025356808,0.001437101,0.00089951983,0.6872903,0.15323,0.04145683,0.1027867,0.00049189175],"about_ca_topic_score_codex":0.012227434,"about_ca_topic_score_gemma":0.015998678,"teacher_disagreement_score":0.012227434,"about_ca_system_score_codex":0.0020501874,"about_ca_system_score_gemma":0.0067128367,"threshold_uncertainty_score":0.036213458},"labels":[],"label_agreement":null},{"id":"W7125939950","doi":"10.1109/ase63991.2025.00251","title":"Who’s to Blame? Rethinking the Brittleness of Automated Web GUI Testing from a Pragmatic Perspective","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland; University of Waterloo","funders":"","keywords":"Automation; Perspective (graphical); Web application; Test (biology); Test case; Software testing; Brittleness","score_opus":0.0263521555170315,"score_gpt":0.30838494502954805,"score_spread":0.28203278951251654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125939950","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.56030846,0.0034423505,0.37811,0.035048828,0.00036115677,0.00032085393,0.0005784785,0.0049755317,0.016854284],"genre_scores_gemma":[0.9415387,0.0003278274,0.055068076,0.0013297831,0.00006433238,0.000064229025,0.0003108795,0.0007325776,0.00056354556],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9519487,0.030647345,0.0021253063,0.0037236451,0.01008412,0.001470932],"domain_scores_gemma":[0.7816958,0.16049078,0.014053959,0.022701308,0.018901266,0.0021569203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05005581,0.000863236,0.0007574979,0.0037122704,0.002118757,0.006817743,0.002747821,0.002171159,0.0018378901],"category_scores_gemma":[0.2223689,0.00097100466,0.0008345832,0.001426218,0.007603667,0.009592277,0.0041225213,0.0033570826,0.0005028994],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000932763,0.0005983667,0.21982294,0.0022036575,0.00039013333,0.005310093,0.054907884,0.03513744,0.04639974,0.11055157,0.020154236,0.50359124],"study_design_scores_gemma":[0.00022439138,0.0010017385,0.10330774,0.0028759441,0.0004922263,0.009174477,0.043374263,0.3494766,0.0391537,0.3314908,0.11870256,0.0007255227],"about_ca_topic_score_codex":0.006031818,"about_ca_topic_score_gemma":0.01205926,"teacher_disagreement_score":0.05005581,"about_ca_system_score_codex":0.002497811,"about_ca_system_score_gemma":0.004461699,"threshold_uncertainty_score":0.2647236},"labels":[],"label_agreement":null},{"id":"W7125982494","doi":"10.1109/ase63991.2025.00211","title":"DRIFT: Debug-based Trace Inference for Firmware Testing","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Firmware; Fuzz testing; Debugging; Concept drift; Binary number; Source code; TRACE (psycholinguistics); Interrupt","score_opus":0.05398414036472179,"score_gpt":0.3325451358902314,"score_spread":0.2785609955255096,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125982494","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018911066,0.00053491467,0.9062811,0.00024180954,0.00014140965,0.00017117977,0.00065559114,0.07131638,0.0017464975],"genre_scores_gemma":[0.50226706,0.00046773563,0.4835356,0.00057860906,0.00012675805,0.0004658859,0.0030276438,0.005031899,0.004498747],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9969024,0.0007878632,0.00021893805,0.0007106295,0.0011138618,0.00026629708],"domain_scores_gemma":[0.9916427,0.004277201,0.00081516046,0.0020003351,0.0009910002,0.00027354498],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028326428,0.0019312064,0.0008395571,0.003412732,0.0006174646,0.0018373731,0.0037042003,0.0016241658,0.004682538],"category_scores_gemma":[0.019446125,0.0010034995,0.0011881788,0.0010657909,0.0014593578,0.0039350986,0.0038955978,0.0021949762,0.0016262509],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011538115,0.0004360296,0.030286724,0.0008987847,0.00030874135,0.0011238679,0.001093362,0.1392265,0.037467353,0.04749131,0.025959427,0.7145542],"study_design_scores_gemma":[0.00012719342,0.0002747231,0.0016541329,0.00011252024,0.00007388101,0.0004009415,0.00011960435,0.912624,0.026174257,0.045151744,0.013207576,0.00007938289],"about_ca_topic_score_codex":0.0035403683,"about_ca_topic_score_gemma":0.004680603,"teacher_disagreement_score":0.004682538,"about_ca_system_score_codex":0.0010998715,"about_ca_system_score_gemma":0.0020985764,"threshold_uncertainty_score":0.015664637},"labels":[],"label_agreement":null},{"id":"W7126241841","doi":"10.1109/iconat66879.2025.11362663","title":"Enhanced Context-Aware Testing for Mobile Applications Using a Hybrid Grammatical Evolution Method","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Genetic programming; Grammatical evolution; Domain (mathematical analysis); Search-based software engineering; Key (lock); Software; Model-based testing; Test case; Test (biology)","score_opus":0.047344349910948914,"score_gpt":0.3659966721439838,"score_spread":0.31865232223303486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7126241841","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09092911,0.0002094121,0.9050505,0.00020765269,0.000031618216,0.00008299761,0.000032581345,0.001144892,0.0023112143],"genre_scores_gemma":[0.68582505,0.00011377543,0.31250337,0.00012842918,0.000013922937,0.000097930075,0.00007766169,0.00010375051,0.001136207],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99951315,0.00017888751,0.000023402963,0.00008314069,0.00016190973,0.00003956358],"domain_scores_gemma":[0.99936265,0.0003602022,0.000055485092,0.000070657836,0.00012655168,0.00002448333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005450077,0.00051009795,0.00040158437,0.00072229485,0.00023163031,0.00042136377,0.0007952686,0.0007192406,0.0007036307],"category_scores_gemma":[0.0024079666,0.0001723844,0.0005632655,0.0003982315,0.0003819495,0.0006362882,0.0006270374,0.0005166523,0.00011550699],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007992833,0.00020327565,0.007317089,0.000119053104,0.00008341699,0.0007853086,0.00034868735,0.5118,0.052940194,0.0113309305,0.0012019409,0.4137901],"study_design_scores_gemma":[0.0000070084857,0.000045772693,0.0004599746,0.000005704894,0.000014401394,0.00011821194,0.000016386492,0.9922769,0.0047093714,0.0016162365,0.0007237907,0.0000060908374],"about_ca_topic_score_codex":0.0022986126,"about_ca_topic_score_gemma":0.0028385401,"teacher_disagreement_score":0.0022986126,"about_ca_system_score_codex":0.00031964606,"about_ca_system_score_gemma":0.0005246369,"threshold_uncertainty_score":0.0045704246},"labels":[],"label_agreement":null},{"id":"W7133004228","doi":"","title":"PathDiff: Systematic Differential Testing Using Symbolic Analysis","year":2022,"lang":"","type":"dissertation","venue":"TSpace","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"University of Toronto","keywords":"Fuzz testing; Path (computing); Symbolic execution; Differential (mechanical device); Process (computing); Software bug; Software","score_opus":0.06123085727894456,"score_gpt":0.37725920008032965,"score_spread":0.3160283428013851,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7133004228","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04140275,0.00018508203,0.94701034,0.00020187415,0.00003797997,0.00011396637,0.00020838705,0.008875863,0.0019637735],"genre_scores_gemma":[0.5265254,0.00016927386,0.470463,0.0001821946,0.000019026864,0.0002309835,0.00053830055,0.000681092,0.0011907674],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964271,0.0013636703,0.00022126103,0.000547062,0.0012166862,0.00022419626],"domain_scores_gemma":[0.9896904,0.0071978397,0.0006949545,0.0015605103,0.0007062708,0.00015012412],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023961004,0.00094744033,0.0006951423,0.0018234659,0.0005629409,0.0010160959,0.0020659547,0.000828621,0.0029890735],"category_scores_gemma":[0.014107333,0.0005509584,0.0011474704,0.0010194222,0.0024036316,0.002319013,0.002476465,0.0013910102,0.00043277317],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008895958,0.00034861546,0.018664192,0.0010287905,0.0002694642,0.00088815,0.0009739059,0.21577181,0.05456463,0.123441525,0.0066446136,0.5765148],"study_design_scores_gemma":[0.00016437394,0.00036275334,0.0012894446,0.00010322509,0.000086299464,0.00042235514,0.00010727838,0.8763794,0.032581754,0.08249156,0.0059617176,0.000049788643],"about_ca_topic_score_codex":0.0020031827,"about_ca_topic_score_gemma":0.0024534247,"teacher_disagreement_score":0.0029890735,"about_ca_system_score_codex":0.0008961128,"about_ca_system_score_gemma":0.002256332,"threshold_uncertainty_score":0.0126719475},"labels":[],"label_agreement":null},{"id":"W7148299376","doi":"10.1504/ijeg.2025.152653","title":"A DQCNN-feedback mechanism-based mobile app testing using MFWKLST-based pattern analysis","year":2025,"lang":"en","type":"article","venue":"International Journal of Electronic Governance","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trusted Positioning (Canada)","funders":"","keywords":"Focus (optics); Graphical user interface; Interface (matter); Mobile device; Quality assurance; User interface; Mobile robot; Mechanism (biology)","score_opus":0.012256202442092553,"score_gpt":0.282056034378821,"score_spread":0.2697998319367284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7148299376","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.108568065,0.00020664254,0.86779976,0.00016695113,0.000099747376,0.00035860462,0.00030672186,0.019365627,0.0031279048],"genre_scores_gemma":[0.6053968,0.00011447868,0.388902,0.00014588091,0.000020403018,0.0002593332,0.0005022135,0.00019127448,0.004467525],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9984518,0.00017893614,0.00011822896,0.00032876508,0.00083398586,0.000088242465],"domain_scores_gemma":[0.9983612,0.00027750793,0.00020742782,0.00024240433,0.00083486974,0.00007660256],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091644947,0.00064120314,0.0005767703,0.0018591356,0.00024866028,0.0005968607,0.0011613806,0.00044676656,0.00263489],"category_scores_gemma":[0.002580828,0.00022073406,0.00037624457,0.0007356401,0.00029881872,0.0008566921,0.00088113925,0.00028412434,0.00088931416],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004932512,0.00024287622,0.008839127,0.00019261763,0.00004530924,0.00033601347,0.00019988097,0.0072403527,0.091862924,0.0012170362,0.0029457011,0.88638484],"study_design_scores_gemma":[0.00011422271,0.00072346115,0.015050363,0.000048153644,0.0000584977,0.0012186776,0.00016571258,0.837278,0.13585639,0.0012675277,0.008138667,0.00008035223],"about_ca_topic_score_codex":0.003852118,"about_ca_topic_score_gemma":0.0043765875,"teacher_disagreement_score":0.003852118,"about_ca_system_score_codex":0.0006171377,"about_ca_system_score_gemma":0.00072560745,"threshold_uncertainty_score":0.008814573},"labels":[],"label_agreement":null},{"id":"W7151951935","doi":"10.1109/aibiec68052.2025.11473598","title":"GAE-HVD: Graph Autoencoder-based Hybrid Vulnerability Detection via Static-Dynamic Program Analysis","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Natural Science Foundation of Guangdong Province; National Natural Science Foundation of China","keywords":"Graph; Program analysis; Computer program; Field (mathematics); Vulnerability (computing)","score_opus":0.01241540303499054,"score_gpt":0.3103491914558713,"score_spread":0.29793378842088075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7151951935","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031258773,0.00029707496,0.94973785,0.0001152218,0.00006425621,0.000060855593,0.0004655133,0.016905904,0.0010945643],"genre_scores_gemma":[0.44402012,0.00021176695,0.5485805,0.00019777165,0.00003742246,0.000113818845,0.0014928817,0.0010243325,0.0043214383],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995696,0.00006454887,0.000015457239,0.00013356685,0.00015803626,0.000058786092],"domain_scores_gemma":[0.99944216,0.00021838564,0.00005812395,0.00014631895,0.0001087819,0.000026240376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035664017,0.0009852604,0.0007757439,0.0016327361,0.00024475754,0.00051111,0.0011320713,0.0006531177,0.0021982454],"category_scores_gemma":[0.0012779788,0.00032493516,0.0006108036,0.0006432664,0.00037868213,0.0008927219,0.0010597277,0.0008960014,0.0008166361],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027211744,0.00022329892,0.0033374934,0.00017242668,0.00017972542,0.00024008488,0.00007969276,0.12781666,0.05897431,0.005180186,0.010299461,0.7932245],"study_design_scores_gemma":[0.000015099022,0.000052313986,0.0009028377,0.000008889081,0.00001976062,0.00010014792,0.000013459543,0.980332,0.0134577025,0.003681293,0.0014036574,0.000012890649],"about_ca_topic_score_codex":0.004179171,"about_ca_topic_score_gemma":0.0072371583,"teacher_disagreement_score":0.004179171,"about_ca_system_score_codex":0.00038962075,"about_ca_system_score_gemma":0.00074102695,"threshold_uncertainty_score":0.008309722},"labels":[],"label_agreement":null},{"id":"W7153576139","doi":"10.1145/3779657.3779658","title":"From Scenario to Code: Structured Prompting for LLM-Based Unit Test Generation","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières; Innovation and Economic Development Trois Rivières","funders":"","keywords":"Unit testing; Test case; Code (set theory); Readability; Test (biology); Java; Software; Relevance (law); Code coverage","score_opus":0.05725744101598945,"score_gpt":0.327343913897419,"score_spread":0.27008647288142956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7153576139","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08212508,0.0005839364,0.787745,0.000974575,0.000180019,0.0006948009,0.0038133818,0.12056717,0.003316112],"genre_scores_gemma":[0.46221757,0.0001726273,0.5207739,0.0007302369,0.00006667853,0.00058020017,0.009472589,0.0041072755,0.0018789015],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.997612,0.0010271942,0.00016631438,0.00057374657,0.00045498903,0.00016578911],"domain_scores_gemma":[0.9923023,0.0046991063,0.0005260489,0.0013198911,0.00085264177,0.00029998523],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017980421,0.0015483106,0.0005909157,0.0016779765,0.00034034398,0.0009910676,0.0017582136,0.0013012537,0.0058069434],"category_scores_gemma":[0.015782392,0.00041971574,0.0009011736,0.0007049177,0.00081273244,0.0015679544,0.002269852,0.0013554655,0.002188234],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014346137,0.00080390734,0.023942066,0.0013411572,0.00010118519,0.0023081324,0.0017929553,0.08504537,0.057176758,0.00964074,0.050930295,0.7654829],"study_design_scores_gemma":[0.00035628106,0.00053646957,0.004040433,0.00019493244,0.00008460049,0.0009924421,0.0006011561,0.8593212,0.06945694,0.035845958,0.028466491,0.00010315019],"about_ca_topic_score_codex":0.0019375337,"about_ca_topic_score_gemma":0.0035792089,"teacher_disagreement_score":0.0058069434,"about_ca_system_score_codex":0.00062662375,"about_ca_system_score_gemma":0.0016294973,"threshold_uncertainty_score":0.019426167},"labels":[],"label_agreement":null},{"id":"W7155389882","doi":"10.13182/t133-49198","title":"A Scenario-driven Automated Testing Framework for Plant Protection System","year":2025,"lang":"","type":"article","venue":"Transactions of the American Nuclear Society","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Automation; Materials testing; System testing; Control system; Systems analysis; Reliability (semiconductor)","score_opus":0.027274014576663563,"score_gpt":0.2735241818869985,"score_spread":0.24625016731033494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7155389882","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007390435,0.00009674196,0.9693944,0.00021691926,0.000030178946,0.0002496473,0.0003952677,0.02025636,0.001970138],"genre_scores_gemma":[0.33034405,0.00016958616,0.6645202,0.00021874826,0.000025702835,0.0003607147,0.001214702,0.0012555099,0.0018907484],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99856794,0.00051588874,0.00009833827,0.00023071388,0.0004487202,0.00013834207],"domain_scores_gemma":[0.997943,0.0010714846,0.00015149784,0.00035042563,0.00032865835,0.00015491975],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021094598,0.0013266078,0.0006863745,0.0011708412,0.0005876852,0.0020062537,0.0031672192,0.0016771363,0.007826768],"category_scores_gemma":[0.004712568,0.00072828674,0.0012505512,0.000385062,0.0009092011,0.00218431,0.0021648824,0.0015571675,0.0011607552],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005754537,0.0005149892,0.0051964666,0.0004601933,0.00025726936,0.0018884522,0.00052926363,0.6859821,0.018292593,0.098283574,0.01593846,0.17208125],"study_design_scores_gemma":[0.00004149558,0.000037546954,0.00017157436,0.000030632895,0.000020362437,0.00016993306,0.000027730652,0.9738514,0.0029340654,0.017397529,0.0052972096,0.00002046767],"about_ca_topic_score_codex":0.008354265,"about_ca_topic_score_gemma":0.01092965,"teacher_disagreement_score":0.008354265,"about_ca_system_score_codex":0.00095818366,"about_ca_system_score_gemma":0.0018580377,"threshold_uncertainty_score":0.026183188},"labels":[],"label_agreement":null},{"id":"W80670519","doi":"10.1007/978-1-4614-4325-4_3","title":"Ordering the Blocks of Designs","year":2012,"lang":"en","type":"book-chapter","venue":"CMS books in mathematics","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science","score_opus":0.0768544269866471,"score_gpt":0.27281935626787385,"score_spread":0.19596492928122675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W80670519","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027357077,0.004382665,0.2982166,0.0017885398,0.0009894281,0.00020371928,0.000537774,0.0016778096,0.6894677],"genre_scores_gemma":[0.082960226,0.006875936,0.3076649,0.0013151168,0.00035649983,0.00044193267,0.0012724085,0.0019564708,0.5971565],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988065,0.00034354685,0.000071759256,0.00021897726,0.00044436942,0.00011479844],"domain_scores_gemma":[0.99853027,0.00058123586,0.000054729335,0.00043311957,0.00032603755,0.00007451208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008086359,0.0013744399,0.00082390377,0.0015649131,0.0016406947,0.004761864,0.0013191848,0.0010014833,0.08910672],"category_scores_gemma":[0.003997057,0.0015528064,0.0011502523,0.0018206639,0.0037247543,0.0063444325,0.0016808833,0.0043733786,0.039129175],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003732185,0.000019398381,0.00009530904,0.0001914703,0.000009432168,0.0000506423,0.0004209361,0.0011379791,0.0011961812,0.8787587,0.029087897,0.08899477],"study_design_scores_gemma":[0.000015930327,0.000036166002,0.0000776299,0.00017696079,0.00001448032,0.000096565724,0.0001375357,0.0015255846,0.0011536129,0.5771403,0.41961107,0.000014116746],"about_ca_topic_score_codex":0.0023727356,"about_ca_topic_score_gemma":0.0035994449,"teacher_disagreement_score":0.08910672,"about_ca_system_score_codex":0.0016293739,"about_ca_system_score_gemma":0.0021751856,"threshold_uncertainty_score":0.2980917},"labels":[],"label_agreement":null},{"id":"W87402359","doi":"10.1007/978-3-642-41533-3_19","title":"Model Checking of UML-RT Models Using Lazy Composition","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Unified Modeling Language; Composition (language); Computer science; Programming language; Model checking; Linguistics; Philosophy; Software","score_opus":0.06041695663218916,"score_gpt":0.2775955526916666,"score_spread":0.21717859605947742,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W87402359","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025866136,0.00009536228,0.96825063,0.00007213338,0.000036036796,0.00008082408,0.00005046736,0.004046241,0.0015020982],"genre_scores_gemma":[0.5199313,0.00022010581,0.47244886,0.00009537797,0.00005088702,0.0001845617,0.00034970068,0.0019994597,0.0047197803],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9943625,0.0019456029,0.00033546798,0.00062531314,0.002194619,0.00053655816],"domain_scores_gemma":[0.9936227,0.0033833156,0.0006513145,0.0015962048,0.00062048243,0.0001259809],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004126188,0.0013147503,0.0014761301,0.0015889035,0.0011735159,0.0021745886,0.0026713184,0.001367542,0.0038883623],"category_scores_gemma":[0.010635002,0.0018755132,0.003787251,0.0009280368,0.001998889,0.004700754,0.0033852807,0.003048274,0.0009310241],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007587797,0.00023938615,0.002894151,0.00059673795,0.00036378918,0.0006976214,0.0009811859,0.63867676,0.043660596,0.17961195,0.001880048,0.12963906],"study_design_scores_gemma":[0.00008095887,0.000076811666,0.00020992586,0.00007391748,0.00012859044,0.000103630955,0.000056233235,0.86771923,0.03030748,0.09874464,0.0024612092,0.00003726515],"about_ca_topic_score_codex":0.004274298,"about_ca_topic_score_gemma":0.007408836,"teacher_disagreement_score":0.004274298,"about_ca_system_score_codex":0.0012304239,"about_ca_system_score_gemma":0.002256943,"threshold_uncertainty_score":0.021821618},"labels":[],"label_agreement":null},{"id":"W965529893","doi":"","title":"Application of model-based approach for testing dynamic systems","year":2011,"lang":"pl","type":"article","venue":"Automatyka / Akademia Górniczo-Hutnicza im. Stanisława Staszica w Krakowie","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Software; Fiscal year; Computer science; Aeronautics; Computer security; Engineering; Reliability engineering; Operations research; Business; Operating system","score_opus":0.07333272393979344,"score_gpt":0.29645735117785654,"score_spread":0.2231246272380631,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W965529893","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004799662,0.00022014168,0.99047536,0.000110843255,0.000026545307,0.000105577,0.00007341777,0.00081649824,0.00337192],"genre_scores_gemma":[0.60870564,0.0006688769,0.38529694,0.00016211835,0.000061149774,0.00055306236,0.00039481474,0.0002313736,0.0039259666],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99849486,0.00050590944,0.00008324789,0.00024881688,0.00056060305,0.00010658743],"domain_scores_gemma":[0.998329,0.0011040985,0.00012060038,0.00017058449,0.0002397302,0.000036056044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013694057,0.001373376,0.0009898914,0.0012834338,0.00057943026,0.0013590539,0.0023736232,0.0013245071,0.004428399],"category_scores_gemma":[0.003992999,0.0006955214,0.0017485581,0.00083552644,0.0010002737,0.0013931865,0.0013002923,0.0015879227,0.0005568861],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005644632,0.000081847036,0.001179831,0.00021186688,0.00009397905,0.0002928931,0.00012184013,0.92902786,0.003044264,0.026587097,0.00064096734,0.038661204],"study_design_scores_gemma":[0.0000081102835,0.000029889508,0.00009298632,0.000015982689,0.000013511797,0.000036520527,0.000017029943,0.98774105,0.0005360044,0.010569652,0.0009334167,0.0000059325353],"about_ca_topic_score_codex":0.0103777945,"about_ca_topic_score_gemma":0.006694053,"teacher_disagreement_score":0.0103777945,"about_ca_system_score_codex":0.0011578839,"about_ca_system_score_gemma":0.0013707235,"threshold_uncertainty_score":0.02063477},"labels":[],"label_agreement":null},{"id":"W989621208","doi":"10.1016/j.scico.2015.07.005","title":"Signature required: Making Simulink data flow and interfaces explicit","year":2015,"lang":"en","type":"article","venue":"Science of Computer Programming","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Signature (topology); Reuse; Documentation; Interface (matter); Generalization; Software engineering; Software; Automotive industry; Comprehension; Program comprehension; Programming language; Software system; Operating system","score_opus":0.08808852585487233,"score_gpt":0.3382831394588602,"score_spread":0.25019461360398787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W989621208","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016628321,0.00006107908,0.9656051,0.00056525535,0.00046292532,0.00013174072,0.00018842985,0.0089818295,0.0073752874],"genre_scores_gemma":[0.58207214,0.00017399433,0.39099452,0.0010764837,0.0002072316,0.00022230815,0.0004848845,0.0050447695,0.01972366],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99588567,0.0008121273,0.00041640317,0.0007295785,0.0015506663,0.0006055719],"domain_scores_gemma":[0.982106,0.004504301,0.0014391394,0.007865093,0.0036191002,0.00046654025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034687747,0.0011184941,0.000958938,0.00075877487,0.0010152343,0.0027646008,0.0028926875,0.0022295022,0.015493593],"category_scores_gemma":[0.022155227,0.0010119848,0.0009412142,0.0007157254,0.002051664,0.008425685,0.0034931959,0.004858386,0.0044297613],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029849005,0.0005458107,0.0030972024,0.000958693,0.00009698018,0.0012718182,0.0011264111,0.02866471,0.09017766,0.4419166,0.016598321,0.4125609],"study_design_scores_gemma":[0.00052166067,0.00046246228,0.00084689795,0.00039383964,0.0002688113,0.00062741473,0.00029226064,0.22788352,0.44654602,0.24538116,0.07661396,0.00016202439],"about_ca_topic_score_codex":0.0010600353,"about_ca_topic_score_gemma":0.0013461001,"teacher_disagreement_score":0.015493593,"about_ca_system_score_codex":0.0010973315,"about_ca_system_score_gemma":0.0034996085,"threshold_uncertainty_score":0.051831245},"labels":[],"label_agreement":null}]}