{"meta":{"query_hash":"8b73c4020eea","filters":{"venue":"IEEE Transactions on Software Engineering"},"cohort_total":213,"direct_labels_cover":1,"predictions_cover":213,"exported":213,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/8b73c4020eea","api":"https://metacan.xera.ac/api/v1/cohort?venue=IEEE+Transactions+on+Software+Engineering"},"results":[{"id":"W1899857608","doi":"10.1109/tse.2015.2431680","title":"Facilitating Coordination between Software Developers: A Study and Techniques for Timely and Efficient Recommendations","year":2015,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"National Science Foundation","keywords":"Computer science; Software engineering; Software; Schedule; Software development; Software project management; Process management; Software construction; Engineering","score_opus":0.041973657220534324,"score_gpt":0.2945908783976059,"score_spread":0.2526172211770716,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1899857608","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27755097,0.00178326,0.70187277,0.003298127,0.00010742497,0.0010883299,0.0003327161,0.0058549773,0.008111357],"genre_scores_gemma":[0.5807394,0.0006462506,0.4153025,0.0001705887,0.00005480064,0.00043215998,0.0003556384,0.00035039664,0.0019482294],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9695877,0.017693575,0.0022742779,0.004335615,0.0052026035,0.00090621854],"domain_scores_gemma":[0.83027154,0.117902964,0.01625291,0.020363905,0.01285437,0.0023542906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020773826,0.0013716507,0.0010091388,0.0045854435,0.00260345,0.0037432506,0.0029282141,0.001874008,0.0018041072],"category_scores_gemma":[0.14197183,0.0014909059,0.000820611,0.0033724254,0.0015330906,0.008288046,0.0024542417,0.0025059995,0.00090850657],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008415842,0.0009429084,0.07088503,0.0011508883,0.00018691606,0.0008115358,0.034007676,0.019014461,0.015587656,0.010856344,0.007265748,0.8384491],"study_design_scores_gemma":[0.00076289166,0.0023980797,0.076923914,0.0015993863,0.0008332223,0.002440135,0.03516954,0.6829632,0.04931013,0.04523197,0.101715244,0.0006523175],"about_ca_topic_score_codex":0.012911481,"about_ca_topic_score_gemma":0.011544011,"teacher_disagreement_score":0.020773826,"about_ca_system_score_codex":0.0022586046,"about_ca_system_score_gemma":0.0051949862,"threshold_uncertainty_score":0.10986382},"labels":[],"label_agreement":null},{"id":"W1963742187","doi":"10.1109/tse.2011.7","title":"Reducing Unauthorized Modification of Digital Objects","year":2011,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Security and Verification in Computing","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Rootkit; Computer science; Operating system; Computer security; Key (lock); Kernel (algebra); Overhead (engineering); File system; Malware; Embedded system","score_opus":0.039293477207764625,"score_gpt":0.23538109404666266,"score_spread":0.19608761683889803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1963742187","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26598936,0.002001231,0.7076419,0.0019095989,0.0002499027,0.0003123346,0.000097560165,0.004043622,0.01775441],"genre_scores_gemma":[0.85285974,0.00065329147,0.13818593,0.00032322912,0.00015221536,0.00008366876,0.00013639996,0.00034358,0.0072619556],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9935341,0.0010313325,0.00040171045,0.00091522915,0.0035572131,0.00056038203],"domain_scores_gemma":[0.95653445,0.009559886,0.004215978,0.024893286,0.004020364,0.0007760292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023740488,0.00085992547,0.0009940676,0.0013987202,0.001363567,0.0024102116,0.0028247382,0.0022194413,0.0024782242],"category_scores_gemma":[0.028644029,0.0005004048,0.000729325,0.00102679,0.002992257,0.008780249,0.005741646,0.0020766715,0.0010172926],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005521121,0.0005955295,0.01864646,0.00087216264,0.00023157257,0.002062269,0.0028353983,0.071207754,0.07646893,0.1462495,0.007900078,0.6723782],"study_design_scores_gemma":[0.00018289086,0.0014636655,0.0067259306,0.00029091287,0.0004345978,0.009031722,0.0013220476,0.46277332,0.20584376,0.231073,0.08065002,0.00020807346],"about_ca_topic_score_codex":0.0008092526,"about_ca_topic_score_gemma":0.00049562496,"teacher_disagreement_score":0.0028247382,"about_ca_system_score_codex":0.00070721423,"about_ca_system_score_gemma":0.001046832,"threshold_uncertainty_score":0.012555361},"labels":[],"label_agreement":null},{"id":"W1984953175","doi":"10.1109/tse.2014.2361131","title":"Replicating and Re-Evaluating the Theory of Relative Defect-Proneness","year":2014,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Polytechnique Montréal; Queen's University","funders":"","keywords":"Computer science; Context (archaeology); Software quality; Replication (statistics); Software; Source code; Software bug; Code review; Quality (philosophy); Software system; Data science; Software development; Software engineering; Statistics; Programming language","score_opus":0.02886620484135756,"score_gpt":0.2784184521015144,"score_spread":0.24955224726015685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1984953175","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21048322,0.032325767,0.6631627,0.042817447,0.010938121,0.0020247009,0.0012418733,0.0013859224,0.035620272],"genre_scores_gemma":[0.81733394,0.0070157624,0.15777966,0.008819358,0.0030212814,0.0016147621,0.0006239606,0.0007286209,0.0030624978],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.82977146,0.11724895,0.0062173503,0.018078059,0.027400767,0.0012833439],"domain_scores_gemma":[0.25004303,0.5635868,0.018999297,0.12691775,0.038364023,0.0020890916],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2518065,0.0030797662,0.0031177802,0.008656621,0.0018425796,0.007104244,0.011806192,0.0041805306,0.0047626756],"category_scores_gemma":[0.62056375,0.00091793336,0.0045053894,0.0054335953,0.015778169,0.015913812,0.005980115,0.009646421,0.0013845625],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012902151,0.0007503312,0.074156605,0.0042402917,0.006902615,0.00079141784,0.014812946,0.04776968,0.0026729032,0.44136235,0.014739631,0.39051098],"study_design_scores_gemma":[0.0006068012,0.0023936925,0.030827474,0.0030912142,0.0019582906,0.00058558903,0.004872724,0.079003386,0.003922858,0.8239341,0.048278697,0.0005251677],"about_ca_topic_score_codex":0.009191635,"about_ca_topic_score_gemma":0.0036856756,"teacher_disagreement_score":0.7481935,"about_ca_system_score_codex":0.006406179,"about_ca_system_score_gemma":0.005826174,"threshold_uncertainty_score":0.9226558},"labels":[],"label_agreement":null},{"id":"W1996993752","doi":"10.1109/tse.2013.28","title":"Early Detection of Collaboration Conflicts and Risks","year":2013,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":78,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Quality (philosophy); Control (management); Source code; Open source; Code (set theory); Software engineering; Risk analysis (engineering); Computer security; Data science; Software; Programming language; Business; Set (abstract data type); Artificial intelligence","score_opus":0.01665302119978797,"score_gpt":0.24331634647688724,"score_spread":0.22666332527709926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1996993752","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60524666,0.0009346738,0.37965217,0.0014716141,0.000087109715,0.00054779893,0.0006290685,0.005038901,0.006392099],"genre_scores_gemma":[0.8503848,0.00013440424,0.14804243,0.0000944984,0.00002037982,0.00011306365,0.00036163323,0.00014006734,0.0007086674],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9887066,0.0029253461,0.0008774013,0.0013851223,0.0054152207,0.00069039554],"domain_scores_gemma":[0.8991385,0.06119677,0.01749227,0.010174309,0.010059381,0.0019387883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00699047,0.0009032242,0.0006570572,0.0034389307,0.0010568554,0.0026907965,0.0020730728,0.0015166346,0.0014781457],"category_scores_gemma":[0.07130098,0.0010637621,0.0005304592,0.0012663908,0.0009568184,0.004116203,0.0037478625,0.0020095627,0.0003721414],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008819402,0.00047280206,0.49819484,0.00084599026,0.00021306833,0.0022940608,0.01118478,0.02339481,0.046761695,0.028181057,0.008632466,0.3789425],"study_design_scores_gemma":[0.00019154076,0.0007787377,0.16768098,0.0005710607,0.00034040815,0.003612338,0.006201289,0.6541431,0.06222674,0.074392766,0.029502587,0.00035852744],"about_ca_topic_score_codex":0.0028356048,"about_ca_topic_score_gemma":0.0031733299,"teacher_disagreement_score":0.00699047,"about_ca_system_score_codex":0.0010617257,"about_ca_system_score_gemma":0.0032263356,"threshold_uncertainty_score":0.036969602},"labels":[],"label_agreement":null},{"id":"W2005680430","doi":"10.1109/tse.2014.2383381","title":"Range Fixes: Interactive Error Resolution for Software Configuration","year":2014,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Key Research and Development Program of China; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Computer science; Range (aeronautics); Constraint (computer-aided design); Software; Simple (philosophy); Theoretical computer science; String (physics); Programming language; Mathematics","score_opus":0.019126007936626662,"score_gpt":0.26072867292940377,"score_spread":0.2416026649927771,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2005680430","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010434081,0.00034010224,0.92918706,0.00013777998,0.00006704961,0.00019813732,0.0007314586,0.05631746,0.0025868681],"genre_scores_gemma":[0.1372999,0.00020733863,0.8520986,0.00012922488,0.00003216625,0.00040731553,0.0018582453,0.006250191,0.0017170439],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9951283,0.001520052,0.00040178082,0.0009679686,0.0016970509,0.00028480584],"domain_scores_gemma":[0.9854344,0.009190155,0.0010786222,0.0033741905,0.0007235769,0.00019897852],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004688331,0.0025993215,0.0009901886,0.0040952493,0.0011017874,0.0023413284,0.0036185484,0.002163877,0.012403033],"category_scores_gemma":[0.027381612,0.0012792124,0.0017567931,0.0019188564,0.0022020463,0.0038595963,0.005157712,0.0021666433,0.003198738],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061468227,0.0002270844,0.008296364,0.0010877366,0.0002093676,0.00074405223,0.0020684307,0.09919992,0.015250532,0.029608607,0.037431985,0.80526125],"study_design_scores_gemma":[0.00026398647,0.00031019835,0.002606599,0.0004974145,0.00013931554,0.0013592592,0.0005889663,0.7919218,0.049242947,0.07973276,0.07305857,0.00027820002],"about_ca_topic_score_codex":0.0022414087,"about_ca_topic_score_gemma":0.0035525984,"teacher_disagreement_score":0.012403033,"about_ca_system_score_codex":0.00079638464,"about_ca_system_score_gemma":0.0011731504,"threshold_uncertainty_score":0.041492283},"labels":[],"label_agreement":null},{"id":"W2032754744","doi":"10.1109/tse.2014.2371458","title":"Guided Mutation Testing for JavaScript Web Applications","year":2015,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Mutation testing; Test suite; Mutation; JavaScript; Programming language; Metric (unit); Web application; Process (computing); Set (abstract data type); Focus (optics); Data mining; Test case; Theoretical computer science; Machine learning; Operating system; Regression analysis","score_opus":0.057086510097033846,"score_gpt":0.2734156176940332,"score_spread":0.21632910759699936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032754744","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6006421,0.000985648,0.38003653,0.0004055768,0.000033013184,0.00031357692,0.00026956876,0.014532301,0.0027818216],"genre_scores_gemma":[0.86092645,0.00018094602,0.13724296,0.000104299324,0.000011804711,0.00011397134,0.00036668626,0.00034656504,0.0007063491],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955011,0.0018758,0.0002812671,0.00041089975,0.0017500209,0.00018089739],"domain_scores_gemma":[0.9876545,0.008244569,0.0015348988,0.001088125,0.0012173029,0.00026057084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025616686,0.00087337056,0.00059683487,0.0018736968,0.00052316394,0.0007575676,0.0013607338,0.00094446517,0.0007162849],"category_scores_gemma":[0.019562904,0.0003424173,0.0005994033,0.00081594154,0.00081962,0.0011771483,0.0008009861,0.00064043055,0.00027123245],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077219855,0.0008483882,0.04184237,0.00050258054,0.00016882634,0.0017401174,0.0006008318,0.2570594,0.14811869,0.007261727,0.0037242603,0.5373605],"study_design_scores_gemma":[0.00010111049,0.00040314195,0.008381547,0.0000688729,0.000070863985,0.0010189987,0.00009174015,0.9105837,0.068339474,0.008333697,0.002563318,0.000043573804],"about_ca_topic_score_codex":0.0028998125,"about_ca_topic_score_gemma":0.003925254,"teacher_disagreement_score":0.0028998125,"about_ca_system_score_codex":0.0007230003,"about_ca_system_score_gemma":0.0012019319,"threshold_uncertainty_score":0.013547599},"labels":[],"label_agreement":null},{"id":"W2053107307","doi":"10.1109/tse.2013.2297712","title":"Automatic Summarization of Bug Reports","year":2014,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":182,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of the Fraser Valley; University of British Columbia","funders":"","keywords":"Automatic summarization; Computer science; Task (project management); Conversation; Software bug; Software; Quality (philosophy); Natural language processing; Information retrieval; Data science; World Wide Web; Software engineering; Programming language","score_opus":0.008384848850770735,"score_gpt":0.2187966960586759,"score_spread":0.21041184720790518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053107307","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53976643,0.004430609,0.39539918,0.0013733484,0.00048392257,0.0013942878,0.010711393,0.040965296,0.005475547],"genre_scores_gemma":[0.6102859,0.0012796634,0.36072332,0.00018953404,0.000397109,0.0007776114,0.021217324,0.0013356162,0.003794006],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99554116,0.0020122086,0.00045516834,0.0007888114,0.0010482243,0.00015440061],"domain_scores_gemma":[0.9578745,0.023564234,0.0049713254,0.0034609116,0.009577771,0.0005512683],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005584284,0.0013514715,0.0010474824,0.0054084975,0.00055270456,0.0016637373,0.0012127068,0.00080411305,0.0018433866],"category_scores_gemma":[0.044444524,0.0004912021,0.00057095475,0.001992469,0.00020929916,0.0018443248,0.0012024656,0.0008204041,0.0012433988],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010538487,0.00032694035,0.020277265,0.0024039813,0.00025207098,0.00062691147,0.0065053715,0.009192116,0.053792942,0.0016344304,0.026800338,0.87713385],"study_design_scores_gemma":[0.0005776219,0.0037143826,0.1258539,0.0011690594,0.0017175485,0.0028095755,0.0075268135,0.5076183,0.18475184,0.012223841,0.15150112,0.00053606863],"about_ca_topic_score_codex":0.001625836,"about_ca_topic_score_gemma":0.002365577,"teacher_disagreement_score":0.005584284,"about_ca_system_score_codex":0.00047597295,"about_ca_system_score_gemma":0.0010457858,"threshold_uncertainty_score":0.02953285},"labels":[],"label_agreement":null},{"id":"W2065698748","doi":"10.1109/tse.2012.8","title":"Guest Editor's Introduction: International Conference on Software Engineering","year":2012,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Special section; Software engineering; Section (typography); Software; Library science; Programming language; Engineering; Operating system; Engineering physics","score_opus":0.017545056608538608,"score_gpt":0.24498084886590438,"score_spread":0.22743579225736577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2065698748","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00008264742,0.021741038,0.00092491234,0.020057783,0.9457429,0.000023250806,0.000136378,0.0001194537,0.011171744],"genre_scores_gemma":[0.0022950883,0.04929873,0.0015473013,0.013248516,0.84411067,0.00008103391,0.0005090065,0.00048393538,0.08842572],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99567753,0.0004605724,0.0005947675,0.00070283114,0.0022141603,0.00035015636],"domain_scores_gemma":[0.98068994,0.002196289,0.0010140163,0.00077499787,0.01193588,0.003388792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043874825,0.003183788,0.0026025043,0.0064454814,0.0018915001,0.008717856,0.0027120742,0.0043604276,0.061132126],"category_scores_gemma":[0.012883532,0.0007371275,0.001859731,0.004397845,0.0011184574,0.0049186093,0.0024986702,0.009169818,0.048692726],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016510245,0.000011957887,0.00006273138,0.00022959012,0.0000075665203,0.000035285055,0.000008729625,0.000062795545,0.00007605555,0.0007768293,0.97518605,0.023525879],"study_design_scores_gemma":[0.000006529764,0.000016588687,0.00024780692,0.0003186137,0.0000110719575,0.0001022817,0.000016611088,0.00011531843,0.0000731272,0.0008067579,0.99827504,0.000010338147],"about_ca_topic_score_codex":0.0015197188,"about_ca_topic_score_gemma":0.0039740494,"teacher_disagreement_score":0.061132126,"about_ca_system_score_codex":0.0024276495,"about_ca_system_score_gemma":0.004167647,"threshold_uncertainty_score":0.20450735},"labels":[],"label_agreement":null},{"id":"W2066455950","doi":"10.1109/tse.2015.2448531","title":"Assessing the Refactorability of Software Clones","year":2015,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":86,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Code refactoring; Computer science; clone (Java method); Maintainability; Programming language; Software maintenance; Code (set theory); Software evolution; Cloning (programming); Source code; Software; Software system; Software engineering; Set (abstract data type); Software construction; Biology; Genetics","score_opus":0.048312374942711414,"score_gpt":0.3005763212471097,"score_spread":0.2522639463043983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2066455950","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9134827,0.001171946,0.08142832,0.000092593546,0.000025979338,0.00017058078,0.0004584902,0.002222262,0.00094707834],"genre_scores_gemma":[0.9164401,0.00034313463,0.08077188,0.00004726043,0.000013490973,0.00008670095,0.00137397,0.00020849792,0.0007148846],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9912062,0.0014448346,0.00087935536,0.0017790439,0.004350164,0.0003403075],"domain_scores_gemma":[0.92036873,0.041429427,0.01699581,0.0060111145,0.014050832,0.001144063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057318173,0.00095933134,0.0008038545,0.007010846,0.00052254327,0.0013133248,0.0013076955,0.0012564708,0.0006271232],"category_scores_gemma":[0.056809045,0.0004437759,0.00096720946,0.0022664517,0.0006983272,0.0019058272,0.0015550931,0.0007334761,0.00035344248],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004598708,0.00023409298,0.6067723,0.0010123986,0.00041835592,0.001008644,0.0020081713,0.024126701,0.04973201,0.0010218157,0.0010556033,0.31214988],"study_design_scores_gemma":[0.00008250737,0.0015417576,0.5830776,0.0004228523,0.0006935825,0.003089601,0.0017423896,0.31052265,0.088195756,0.003161649,0.007234302,0.00023538394],"about_ca_topic_score_codex":0.0034147662,"about_ca_topic_score_gemma":0.0042250105,"teacher_disagreement_score":0.007010846,"about_ca_system_score_codex":0.00070194196,"about_ca_system_score_gemma":0.0010225524,"threshold_uncertainty_score":0.030313134},"labels":[],"label_agreement":null},{"id":"W2067617772","doi":"10.1109/tse.2014.2363479","title":"Instance Generator and Problem Representation to Improve Object Oriented Code Coverage","year":2014,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":86,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Unit testing; Test case; Programming language; Code generation; Java; Code coverage; Object-oriented programming; Data structure; Source code; Generator (circuit theory); Software; Theoretical computer science; Machine learning; Operating system","score_opus":0.009565537506992744,"score_gpt":0.23054314654182123,"score_spread":0.2209776090348285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2067617772","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061452817,0.0002558603,0.9303147,0.00041040045,0.000043555832,0.0003015467,0.00027200987,0.0042056898,0.0027433787],"genre_scores_gemma":[0.30875137,0.00014666683,0.6867939,0.00019079843,0.00004021052,0.000581838,0.0011326323,0.00071254605,0.0016499636],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980094,0.00091809797,0.00013072437,0.00026941416,0.00052619755,0.00014613442],"domain_scores_gemma":[0.99071336,0.007262501,0.00036670253,0.00081413565,0.00070327363,0.00013998128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023281644,0.0010175756,0.00083197403,0.0015311969,0.00033179036,0.0010157257,0.0016761124,0.0012188375,0.0034384422],"category_scores_gemma":[0.016021011,0.00047048883,0.0011986118,0.0011815666,0.00088339136,0.0014141997,0.0016391369,0.0012809803,0.0005158788],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003591938,0.0005333858,0.006147843,0.0005472008,0.000110511515,0.00045382237,0.00041982537,0.63214445,0.012555314,0.037743483,0.008172878,0.30081216],"study_design_scores_gemma":[0.000078819634,0.000060693546,0.00024681306,0.000020275153,0.000025980731,0.000088635454,0.000032182463,0.98330265,0.004014054,0.009934347,0.0021875813,0.000007931818],"about_ca_topic_score_codex":0.0014607677,"about_ca_topic_score_gemma":0.0017441,"teacher_disagreement_score":0.0034384422,"about_ca_system_score_codex":0.00075841846,"about_ca_system_score_gemma":0.0013874777,"threshold_uncertainty_score":0.012312591},"labels":[],"label_agreement":null},{"id":"W2068493323","doi":"10.1109/tse.2011.56","title":"Size-Constrained Regression Test Case Selection Using Multicriteria Optimization","year":2011,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Test suite; Test case; Consistency (knowledge bases); Integer programming; Mathematical optimization; Task (project management); Constraint (computer-aided design); Selection (genetic algorithm); Regression testing; Greedy algorithm; Linear programming; Relaxation (psychology); Software; Algorithm; Machine learning; Regression analysis; Artificial intelligence; Software system; Mathematics","score_opus":0.03160363037280706,"score_gpt":0.24695615889016168,"score_spread":0.21535252851735462,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2068493323","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042217884,0.0002500273,0.9535501,0.00018956552,0.00001864077,0.00027929642,0.00005645239,0.0011299348,0.0023079638],"genre_scores_gemma":[0.50886685,0.00013158054,0.48781067,0.00019684482,0.000035145255,0.000607582,0.0003522329,0.00030591027,0.0016932025],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99636984,0.001880121,0.000155738,0.00046025356,0.00080767577,0.000326385],"domain_scores_gemma":[0.98945516,0.008016248,0.0009346336,0.00050066254,0.00092964614,0.00016358956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003340231,0.0017932216,0.0021227065,0.0025780003,0.00045570367,0.0009586763,0.0019414583,0.0010370733,0.0022639323],"category_scores_gemma":[0.012145688,0.00081527745,0.0013888077,0.0014915229,0.0008785232,0.0010260463,0.00104371,0.0011201613,0.00039728318],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018284851,0.00015764721,0.0014335392,0.00008623061,0.000092613016,0.00010823306,0.000049978273,0.88577604,0.0054393634,0.0028116147,0.0008575861,0.10300437],"study_design_scores_gemma":[0.000030175184,0.00006433009,0.00021401522,0.0000059557965,0.00001546945,0.000025702788,0.000008331274,0.99693537,0.0013848626,0.0011398932,0.00016995223,0.000005929534],"about_ca_topic_score_codex":0.0040839,"about_ca_topic_score_gemma":0.004639899,"teacher_disagreement_score":0.0040839,"about_ca_system_score_codex":0.0012796273,"about_ca_system_score_gemma":0.0018423954,"threshold_uncertainty_score":0.017665029},"labels":[],"label_agreement":null},{"id":"W2075639135","doi":"10.1109/tse.2014.2354043","title":"Customizing the Representation Capabilities of Process Models: Understanding the Effects of Perceived Modeling Impediments","year":2014,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Business Process Modeling and Analysis","field":"Business, Management and Accounting","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Process (computing); Computer science; Process modeling; Representation (politics); Variety (cybernetics); Process mining; Design process; Data science; Process management; Work in process; Management science; Business process modeling; Artificial intelligence; Engineering; Business process","score_opus":0.023471896139688447,"score_gpt":0.22315438599892637,"score_spread":0.19968248985923792,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2075639135","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9676407,0.00021778785,0.026952447,0.0009988791,0.000016714252,0.00014683545,0.000028395883,0.00023563893,0.0037625972],"genre_scores_gemma":[0.99292374,0.00006638076,0.0067503992,0.000054752352,0.000006327989,0.000037653535,0.000029632296,0.000031991116,0.00009922458],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9478618,0.03187791,0.005316297,0.002777602,0.010076434,0.002090017],"domain_scores_gemma":[0.45276633,0.43529543,0.054981507,0.03388985,0.01919526,0.0038717287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.050002858,0.0011434768,0.00044243204,0.0021213384,0.0014695786,0.0062263682,0.0016857282,0.0020319053,0.0014641506],"category_scores_gemma":[0.3157939,0.0012196103,0.0011666468,0.001218527,0.0034402024,0.011252553,0.0043510846,0.0037146818,0.00016181929],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002003545,0.0018641463,0.5376497,0.0017533374,0.00080030406,0.0018796452,0.13672805,0.062194876,0.04387476,0.028959973,0.0009879092,0.18130383],"study_design_scores_gemma":[0.0004187279,0.0041407305,0.47811505,0.0023560016,0.0012109951,0.002899112,0.10232012,0.3053331,0.031553168,0.05301138,0.017690584,0.00095103873],"about_ca_topic_score_codex":0.005381466,"about_ca_topic_score_gemma":0.0035308662,"teacher_disagreement_score":0.050002858,"about_ca_system_score_codex":0.0030088513,"about_ca_system_score_gemma":0.0021148329,"threshold_uncertainty_score":0.26444358},"labels":[],"label_agreement":null},{"id":"W2085597081","doi":"10.1109/tse.2013.27","title":"The Impact of Classifier Configuration and Classifier Combination on Bug Localization","year":2013,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Classifier (UML); Computer science; Artificial intelligence; Machine learning; Random subspace method; Source code; Probabilistic classification; Quadratic classifier; Data mining; Pattern recognition (psychology); Support vector machine; Naive Bayes classifier; Programming language","score_opus":0.013915283300651619,"score_gpt":0.24821500259989007,"score_spread":0.23429971929923846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2085597081","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95140713,0.010654561,0.025231441,0.0015126854,0.00068819826,0.00058282184,0.00079145556,0.0045470865,0.0045846426],"genre_scores_gemma":[0.98006964,0.0007411769,0.016451383,0.0003187121,0.00020003102,0.00019421501,0.0010778852,0.00032570615,0.00062128954],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9662212,0.014309876,0.004156333,0.00836386,0.004861466,0.0020871838],"domain_scores_gemma":[0.81041646,0.14270714,0.008972839,0.02130262,0.012273375,0.0043274667],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.029137626,0.0049952907,0.0037960983,0.0058547407,0.0029386324,0.005440454,0.0032457644,0.004519301,0.0013052499],"category_scores_gemma":[0.120930776,0.0019142148,0.0020257016,0.0048329155,0.0026089463,0.011590029,0.0032470978,0.0042903135,0.0014245462],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008782708,0.0029235824,0.29555288,0.0012875304,0.0031354604,0.0011441316,0.0011435627,0.21447873,0.023629395,0.00096890936,0.012393992,0.43455905],"study_design_scores_gemma":[0.0014538725,0.015429636,0.14513414,0.0008166039,0.006380394,0.0055744858,0.0036038328,0.72750235,0.07174043,0.008071773,0.013088311,0.0012041584],"about_ca_topic_score_codex":0.0043016137,"about_ca_topic_score_gemma":0.0037196036,"teacher_disagreement_score":0.029137626,"about_ca_system_score_codex":0.0019843348,"about_ca_system_score_gemma":0.0022641628,"threshold_uncertainty_score":0.15409636},"labels":[],"label_agreement":null},{"id":"W2089113163","doi":"10.1109/tse.2013.29","title":"Usability through Software Design","year":2013,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Usability and User Interface Design","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Usability; Computer science; Usability inspection; Component-based usability testing; Usability engineering; Cognitive walkthrough; Usability goals; Software engineering; Usability lab; Heuristic evaluation; Software development; Web usability; Software; Human–computer interaction; Programming language","score_opus":0.029666864984378358,"score_gpt":0.2302247028682714,"score_spread":0.20055783788389303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2089113163","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008615964,0.022797018,0.8074149,0.029088147,0.0012768038,0.0025637,0.00022633882,0.0012798337,0.12673728],"genre_scores_gemma":[0.17616165,0.022408867,0.7738675,0.005369018,0.00077883084,0.005344418,0.00035154022,0.0008011662,0.014917104],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.86825144,0.100159004,0.006368163,0.0061376155,0.017661268,0.0014225583],"domain_scores_gemma":[0.9148117,0.06618291,0.0021903506,0.007185713,0.008111972,0.001517359],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06761111,0.0038766377,0.002810593,0.0112109985,0.0043788925,0.021766698,0.003863868,0.0054174005,0.0057236184],"category_scores_gemma":[0.06415601,0.0017052126,0.0020917093,0.005112878,0.038337033,0.017128026,0.011891182,0.007597808,0.0021435134],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004861931,0.00018975364,0.0014579919,0.0035886625,0.00015700524,0.00019993985,0.019219058,0.0031811579,0.0013441266,0.8309084,0.010002977,0.12970221],"study_design_scores_gemma":[0.00009234127,0.00021273165,0.00078123296,0.004005667,0.000108781685,0.00043876996,0.005320643,0.004223253,0.0019262183,0.7412009,0.24158458,0.00010486694],"about_ca_topic_score_codex":0.0053020585,"about_ca_topic_score_gemma":0.002713709,"teacher_disagreement_score":0.06761111,"about_ca_system_score_codex":0.008821234,"about_ca_system_score_gemma":0.0145442765,"threshold_uncertainty_score":0.357566},"labels":[],"label_agreement":null},{"id":"W2095974519","doi":"10.1109/tse.2010.32","title":"Assessing, Comparing, and Combining State Machine-Based Testing and Structural Testing: A Series of Experiments","year":2010,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Test strategy; Non-regression testing; Software performance testing; White-box testing; Machine learning; Reliability engineering; Model-based testing; Software testing; Risk-based testing; Software; Series (stratigraphy); Code coverage; State (computer science); Keyword-driven testing; Software reliability testing; Data mining; Test case; Software quality; Algorithm; Software system; Software development; Engineering; Software construction","score_opus":0.034421962619943035,"score_gpt":0.2701052537879602,"score_spread":0.23568329116801714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095974519","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97860765,0.00020046812,0.017885275,0.000085902466,0.000052065774,0.0013315949,0.00028820123,0.0002922184,0.0012566929],"genre_scores_gemma":[0.9555268,0.00020497966,0.03990018,0.00011580654,0.000038802027,0.0023806794,0.0006231075,0.00008808241,0.0011214742],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9917631,0.004088154,0.00090368197,0.0011800597,0.0015791826,0.00048579942],"domain_scores_gemma":[0.9114702,0.07438717,0.0038898138,0.0058213347,0.003445126,0.0009864353],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007070511,0.0018822305,0.001360801,0.0012413356,0.0004689528,0.0008518606,0.002129822,0.0017532419,0.0021390193],"category_scores_gemma":[0.03658542,0.00067412434,0.0011200706,0.00089196773,0.001152671,0.0020088542,0.0011708707,0.0012317771,0.00035883562],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.05284751,0.0768852,0.02951685,0.0034656643,0.0016389025,0.0007389088,0.003403259,0.2880326,0.23557474,0.0056206253,0.002859672,0.29941615],"study_design_scores_gemma":[0.0085663,0.25125974,0.03511932,0.00023322666,0.0013912476,0.0006388863,0.001437558,0.42150342,0.26246235,0.008947087,0.0079782745,0.00046254008],"about_ca_topic_score_codex":0.0016843845,"about_ca_topic_score_gemma":0.001675944,"teacher_disagreement_score":0.007070511,"about_ca_system_score_codex":0.00105869,"about_ca_system_score_gemma":0.0010838535,"threshold_uncertainty_score":0.037392914},"labels":[],"label_agreement":null},{"id":"W2097750323","doi":"10.1109/tse.2008.26","title":"Asking and Answering Questions during a Programming Change Task","year":2008,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":314,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada; University of Alberta; Vrije Universiteit Brussel; University of Washington","keywords":"Computer science; Programmer; Ask price; Task (project management); Key (lock); Programming language; Question answering; Code (set theory); Software engineering; World Wide Web; Data science; Information retrieval","score_opus":0.022144955307222974,"score_gpt":0.23619539299239536,"score_spread":0.2140504376851724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2097750323","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9550355,0.00032917765,0.036767155,0.0016592955,0.00003826265,0.00046819478,0.00028252616,0.0010382457,0.0043815514],"genre_scores_gemma":[0.9519398,0.00026437122,0.043155007,0.00090824364,0.000048127895,0.00051241874,0.000560074,0.00028482915,0.0023269735],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9675163,0.023655517,0.001625967,0.002573856,0.0031507933,0.0014775818],"domain_scores_gemma":[0.6370504,0.32284904,0.013663167,0.012380022,0.009353153,0.0047043124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019963067,0.000797919,0.0010076755,0.0022958203,0.0026854621,0.0032947452,0.0021485249,0.0041041854,0.0029812397],"category_scores_gemma":[0.18126258,0.0011956067,0.0006895764,0.0014714572,0.0021204131,0.0055202525,0.003897749,0.0032844313,0.0010571798],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012317931,0.00076429924,0.18357284,0.0011807234,0.00013655235,0.0018620827,0.62082964,0.0015760697,0.044308584,0.0033806784,0.006171025,0.13498572],"study_design_scores_gemma":[0.00045278118,0.0037935546,0.30222824,0.0010819273,0.00034786077,0.0060992027,0.44809082,0.050607804,0.035533577,0.016047873,0.13485453,0.00086195825],"about_ca_topic_score_codex":0.0030734115,"about_ca_topic_score_gemma":0.0032775623,"teacher_disagreement_score":0.019963067,"about_ca_system_score_codex":0.0011955246,"about_ca_system_score_gemma":0.0016340293,"threshold_uncertainty_score":0.1055761},"labels":[],"label_agreement":null},{"id":"W2099069768","doi":"10.1109/tse.2005.106","title":"Analyzing the evolutionary history of the logical design of object-oriented software","year":2005,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Software evolution; Unified Modeling Language; Object-oriented design; Software system; Class (philosophy); Programming language; Inheritance (genetic algorithm); Sequence diagram; Class diagram; Abstraction; Object-oriented programming; Sequence (biology); Software; Software engineering; Artificial intelligence; Software construction","score_opus":0.02065955642586971,"score_gpt":0.22516811738749967,"score_spread":0.20450856096162995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099069768","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7357845,0.0011515311,0.25726047,0.0005285609,0.000016967851,0.000114083596,0.00021825572,0.00014745645,0.004778227],"genre_scores_gemma":[0.8466096,0.00066548074,0.1508315,0.000060888222,0.00001217632,0.000073524265,0.00047795288,0.000053832657,0.0012150094],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99850225,0.00063524576,0.00009252697,0.00020623313,0.00050191156,0.00006183554],"domain_scores_gemma":[0.9926267,0.0035163707,0.0013162781,0.00092207664,0.0014496269,0.00016882762],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032914602,0.000182372,0.00017420598,0.0030481,0.00076851377,0.0013621825,0.0004493118,0.00043592346,0.0005747259],"category_scores_gemma":[0.01522747,0.0003478626,0.00029683538,0.002226571,0.0009479161,0.0027116616,0.00059444393,0.00072121114,0.00010917307],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022457707,0.00022726982,0.3204443,0.00040068192,0.00016468343,0.00063978374,0.0082708625,0.041073862,0.020456007,0.11186485,0.00080422993,0.495429],"study_design_scores_gemma":[0.000058432106,0.0006254491,0.43315938,0.0004084705,0.00025291066,0.0017994716,0.0046213726,0.34024107,0.038779266,0.12944877,0.05042516,0.00018021766],"about_ca_topic_score_codex":0.0035250813,"about_ca_topic_score_gemma":0.0052732914,"teacher_disagreement_score":0.0035250813,"about_ca_system_score_codex":0.0016821922,"about_ca_system_score_gemma":0.0011894187,"threshold_uncertainty_score":0.01740712},"labels":[],"label_agreement":null},{"id":"W2099441126","doi":"10.1109/tse.2003.1214327","title":"General test result checking with log file analysis","year":2003,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":95,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Random testing; Formalism (music); Programming language; Unit testing; Operating system; Software; Test case","score_opus":0.011147187604348262,"score_gpt":0.21500120278407897,"score_spread":0.2038540151797307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099441126","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007896787,0.00007660389,0.98087364,0.000092103066,0.00001678504,0.0001476252,0.00012692307,0.00981705,0.00095251703],"genre_scores_gemma":[0.31692448,0.00016372801,0.67774165,0.00025091952,0.000082385974,0.0006486533,0.00097246474,0.0014334787,0.0017823555],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9863593,0.0045703594,0.0013133332,0.0015297352,0.005437585,0.00078964775],"domain_scores_gemma":[0.96384144,0.017055323,0.0031446638,0.012552184,0.0030804514,0.00032589884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005819703,0.0016131802,0.0012983641,0.0032776836,0.00071523216,0.0032212331,0.0040414254,0.0016840895,0.0051211435],"category_scores_gemma":[0.036806498,0.00080929877,0.001706109,0.00183454,0.0033134886,0.006775377,0.0033063265,0.0019516095,0.001431096],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018721181,0.0009540087,0.017678656,0.00185562,0.00030088646,0.0018022475,0.0010512437,0.111706346,0.061787598,0.18405148,0.014171024,0.6027688],"study_design_scores_gemma":[0.00025620218,0.0006021031,0.0023314634,0.00032876688,0.00019285691,0.002089937,0.00013428011,0.70148623,0.122601226,0.1508023,0.018990496,0.00018417792],"about_ca_topic_score_codex":0.0021805032,"about_ca_topic_score_gemma":0.0016057021,"teacher_disagreement_score":0.005819703,"about_ca_system_score_codex":0.0013185618,"about_ca_system_score_gemma":0.002989001,"threshold_uncertainty_score":0.030777931},"labels":[],"label_agreement":null},{"id":"W2100310705","doi":"10.1109/tse.2007.70747","title":"API-Evolution Support with Diff-CatchUp","year":2007,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":157,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Component (thermodynamics); Reuse; Software engineering; Context (archaeology); Component-based software engineering; Generality; Software evolution; Software development; Software; Programming language; Software construction","score_opus":0.014790567950770252,"score_gpt":0.23751267404951754,"score_spread":0.2227221060987473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100310705","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09339323,0.00032257117,0.7016426,0.00055408414,0.00011776899,0.00051058986,0.00026252252,0.19734485,0.005851712],"genre_scores_gemma":[0.46104276,0.00022079427,0.5115122,0.00086509937,0.00007982428,0.0004937854,0.0014268894,0.015476384,0.008882228],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9966979,0.0005686049,0.0004081292,0.0006665148,0.0013632711,0.00029556564],"domain_scores_gemma":[0.98180777,0.0060906843,0.0013518503,0.008611391,0.0014344888,0.000703844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004742615,0.0016465873,0.00093624485,0.0013791412,0.00082749466,0.001981919,0.0069577675,0.0023122411,0.004218207],"category_scores_gemma":[0.019673012,0.001732153,0.0010696443,0.0008565644,0.0013562267,0.0075146053,0.008729583,0.0033164616,0.0011444183],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019544244,0.0010303543,0.020866288,0.00081674644,0.00019466256,0.0043206415,0.003607457,0.01416278,0.071961425,0.029146748,0.031200826,0.82073766],"study_design_scores_gemma":[0.00091051566,0.00085974974,0.011656595,0.0002711894,0.00034754095,0.00643882,0.00075064803,0.5594849,0.2374152,0.048575886,0.13265318,0.0006358279],"about_ca_topic_score_codex":0.0021337261,"about_ca_topic_score_gemma":0.0021879193,"teacher_disagreement_score":0.0069577675,"about_ca_system_score_codex":0.00070229825,"about_ca_system_score_gemma":0.0011568861,"threshold_uncertainty_score":0.025081635},"labels":[],"label_agreement":null},{"id":"W2101198558","doi":"10.1109/32.877844","title":"Advanced exception handling mechanisms","year":2000,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":76,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Exception handling; Computer science; Programming language; Concurrency; Feature (linguistics); Concurrency control; Object (grammar); Artificial intelligence","score_opus":0.01509469912932844,"score_gpt":0.2357682562586422,"score_spread":0.22067355712931375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2101198558","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013310759,0.002401682,0.934081,0.0019836717,0.0012432203,0.00060052617,0.00027217244,0.016422352,0.029684573],"genre_scores_gemma":[0.21652499,0.0035623836,0.70521957,0.0033200695,0.0020831344,0.0008310138,0.00075442484,0.0027763692,0.06492802],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9877116,0.0020225793,0.0024281088,0.0014759882,0.005337646,0.001024111],"domain_scores_gemma":[0.96899366,0.0061020763,0.0038495595,0.012343508,0.007576423,0.0011348389],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014245454,0.0013512811,0.0013046501,0.0024733618,0.0015258539,0.007367872,0.008743908,0.0030150611,0.010601033],"category_scores_gemma":[0.028730497,0.0010360483,0.0020466538,0.0017718239,0.0023386246,0.010466606,0.007748271,0.0045796623,0.0053277155],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038841192,0.00025922692,0.002848365,0.0017879681,0.0001944099,0.0009739698,0.002033607,0.006526626,0.023631642,0.65537465,0.028874774,0.2771063],"study_design_scores_gemma":[0.000293273,0.00054844405,0.0013692958,0.0012304984,0.00040799446,0.002882338,0.0003330557,0.038093746,0.027050449,0.2396857,0.6877217,0.00038338208],"about_ca_topic_score_codex":0.000550011,"about_ca_topic_score_gemma":0.00053678115,"teacher_disagreement_score":0.014245454,"about_ca_system_score_codex":0.0009927304,"about_ca_system_score_gemma":0.0028374807,"threshold_uncertainty_score":0.075338066},"labels":[],"label_agreement":null},{"id":"W2102418068","doi":"10.1109/tse.2002.1049402","title":"Timed Wp-method: testing real-time systems","year":2002,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Nondeterministic algorithm; Correctness; Timed automaton; Automaton; System under test; Finite-state machine; Fault coverage; Real-time operating system; Test case; Real-time computing; Distributed computing; Algorithm; Theoretical computer science; Embedded system","score_opus":0.026720613989180138,"score_gpt":0.23424799388672066,"score_spread":0.20752737989754053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102418068","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017787261,0.000079057194,0.97820765,0.000040283998,0.000022459924,0.00016110128,0.00017809871,0.0026322675,0.00089181546],"genre_scores_gemma":[0.3384065,0.00013720409,0.6577503,0.00007711513,0.000024796906,0.0005239425,0.0007294917,0.000625727,0.0017248413],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99785846,0.0007516762,0.00017018586,0.00044345448,0.000670169,0.00010597851],"domain_scores_gemma":[0.9958978,0.0025394445,0.0003507762,0.00076986034,0.0003586235,0.00008337678],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011310381,0.0012176683,0.00047070326,0.0007330701,0.00017078125,0.0004943058,0.0021824485,0.00075739034,0.003019999],"category_scores_gemma":[0.0061264904,0.00033055682,0.00072343455,0.00063532987,0.00076387846,0.0010477833,0.000606149,0.0006283829,0.0004527835],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007433625,0.00036308254,0.010028492,0.0021642926,0.00033041986,0.0017609013,0.0005391372,0.32070854,0.13414524,0.039218828,0.004840839,0.48515695],"study_design_scores_gemma":[0.00018702217,0.0004940294,0.0019521584,0.00010910156,0.00010386623,0.0011929697,0.00009255286,0.8449922,0.118267946,0.024052521,0.008509327,0.00004621792],"about_ca_topic_score_codex":0.0013722988,"about_ca_topic_score_gemma":0.00087739347,"teacher_disagreement_score":0.003019999,"about_ca_system_score_codex":0.00038804614,"about_ca_system_score_gemma":0.0008638132,"threshold_uncertainty_score":0.010102868},"labels":[],"label_agreement":null},{"id":"W2103640219","doi":"10.1109/tse.2005.28","title":"Using origin analysis to detect merging and splitting of source code entities","year":2005,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":252,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Context (archaeology); Source code; Code (set theory); Abstraction; Programming language; Software; Plan (archaeology); Software evolution; Software engineering; Database; Theoretical computer science; Software system; Data mining; Software construction","score_opus":0.023893764511130126,"score_gpt":0.2743728504647013,"score_spread":0.25047908595357116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2103640219","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33771923,0.00064919505,0.637685,0.00033787618,0.000060806604,0.00022895577,0.0007312484,0.019125165,0.0034625505],"genre_scores_gemma":[0.64331436,0.0002477558,0.35109803,0.000114227776,0.000043857417,0.00012345998,0.0013101421,0.0017540533,0.0019940583],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9959111,0.0008185263,0.00041816448,0.0008737332,0.0017001289,0.00027837566],"domain_scores_gemma":[0.97344524,0.011054456,0.005633986,0.0047767805,0.004662114,0.00042734374],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038887241,0.00060332543,0.0006607799,0.0065161865,0.00088273553,0.0020654867,0.0018140787,0.001224463,0.001415082],"category_scores_gemma":[0.019867152,0.0005309813,0.00090425374,0.003157291,0.0013739556,0.0036548385,0.0029741076,0.0013458348,0.0004939389],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001238193,0.00025422336,0.25722247,0.0008503843,0.0002978372,0.0040342645,0.01380366,0.010994925,0.094051644,0.02901814,0.0051191533,0.58311504],"study_design_scores_gemma":[0.0002247944,0.0006805702,0.107957356,0.00026817364,0.0007655524,0.0053645596,0.0032623126,0.44403163,0.34021935,0.038992427,0.05782363,0.00040969584],"about_ca_topic_score_codex":0.003101582,"about_ca_topic_score_gemma":0.0029623036,"teacher_disagreement_score":0.0065161865,"about_ca_system_score_codex":0.0009171635,"about_ca_system_score_gemma":0.0011919173,"threshold_uncertainty_score":0.020565808},"labels":[],"label_agreement":null},{"id":"W2104994910","doi":"10.1109/tse.2012.71","title":"Trustrace: Mining Software Repositories to Improve the Accuracy of Requirement Traceability Links","year":2012,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":105,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Traceability; Computer science; Requirements traceability; Precision and recall; Source code; Software; Data mining; Software engineering; Software development; Database; Information retrieval; Programming language; Requirement","score_opus":0.020912743124051024,"score_gpt":0.27159741001598686,"score_spread":0.25068466689193586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104994910","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24205667,0.0028220494,0.70421004,0.0007849833,0.000123654,0.00045568417,0.0030681002,0.043277077,0.0032017329],"genre_scores_gemma":[0.63313824,0.0007535007,0.35141376,0.00017558479,0.000054871714,0.00027522363,0.010684615,0.0010584494,0.0024457583],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.991542,0.001645853,0.0010442545,0.0013810429,0.0040021017,0.00038475555],"domain_scores_gemma":[0.95435274,0.017463408,0.007647164,0.009763036,0.010247598,0.0005260101],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006811192,0.0018642737,0.0015505361,0.012617023,0.00094947766,0.00257971,0.0027703426,0.0013788294,0.0007195793],"category_scores_gemma":[0.058251232,0.0007106504,0.0015270152,0.006128544,0.0006426151,0.0069535575,0.0030546826,0.0014379286,0.00092555373],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005235771,0.0008655627,0.08688509,0.0014889684,0.0005211903,0.000842177,0.0017727795,0.061769076,0.021602958,0.004016467,0.013338029,0.8063741],"study_design_scores_gemma":[0.000086900276,0.0005223567,0.019618638,0.00016542485,0.00024206971,0.00087020313,0.00070098386,0.91096,0.048681796,0.008971738,0.009039544,0.00014038033],"about_ca_topic_score_codex":0.00931622,"about_ca_topic_score_gemma":0.008612623,"teacher_disagreement_score":0.012617023,"about_ca_system_score_codex":0.00092570274,"about_ca_system_score_gemma":0.0023928757,"threshold_uncertainty_score":0.03602141},"labels":[],"label_agreement":null},{"id":"W2105246013","doi":"10.1109/tse.2002.1158289","title":"Ethical issues in empirical studies of software engineering","year":2002,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":190,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Health and Medical Research Council; Medical Research Council; National Research Council Canada; U.S. Department of Health and Human Services; National Institutes of Health; Public Works and Government Services Canada; Canadian Psychological Association","keywords":"Empirical research; Computer science; Social software engineering; Software engineering; Popularity; Software Engineering Process Group; Software development; Software requirements; Software peer review; Personal software process; Software; Management science; Data science; Software construction; Engineering ethics; Engineering","score_opus":0.05595567562007852,"score_gpt":0.3199783701972545,"score_spread":0.264022694577176,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105246013","genre_codex":"methods","genre_gemma":"empirical","domain_codex":"methods","domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":"methods","prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03574403,0.03808928,0.439704,0.34727132,0.00906069,0.006020498,0.00025893777,0.00026313338,0.12358805],"genre_scores_gemma":[0.50068843,0.01846936,0.3172547,0.12581588,0.008346492,0.017695641,0.00021465516,0.00032726896,0.011187562],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.30115083,0.6081276,0.024566788,0.0104304245,0.052676752,0.0030475657],"domain_scores_gemma":[0.16343462,0.72128344,0.03067684,0.04677941,0.033757932,0.00406771],"candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.4185021,0.0014129147,0.002796659,0.005083118,0.012825445,0.016176123,0.0038559057,0.014856556,0.002913034],"category_scores_gemma":[0.61155987,0.0016576989,0.0011353245,0.007767622,0.06434058,0.017949274,0.011470998,0.020015236,0.0010620625],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008946834,0.00013928408,0.0029046063,0.0011104214,0.00009299758,0.0005726039,0.053810887,0.0005371869,0.0004678984,0.8818937,0.014190977,0.044189952],"study_design_scores_gemma":[0.00012901887,0.00020941309,0.0021746869,0.0059072156,0.0000737396,0.0007713637,0.02316668,0.0020420342,0.0010105559,0.77500975,0.18938291,0.0001226024],"about_ca_topic_score_codex":0.001344341,"about_ca_topic_score_gemma":0.0022222097,"teacher_disagreement_score":0.9851434,"about_ca_system_score_codex":0.0057713334,"about_ca_system_score_gemma":0.020424109,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2107024044","doi":"10.1109/tse.2006.38","title":"On the value of static analysis for fault detection in software","year":2006,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":281,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nortel (Canada)","funders":"North Carolina State University; National Science Foundation","keywords":"Static analysis; Computer science; Fault detection and isolation; Software; Software quality; Reliability engineering; Software bug; Programmer; Software reliability testing; Fault (geology); Static program analysis; Data mining; Software development; Embedded system; Operating system; Artificial intelligence; Programming language; Engineering","score_opus":0.011465499236457887,"score_gpt":0.23584611802258723,"score_spread":0.22438061878612933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2107024044","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44364768,0.0064586787,0.53238446,0.002100173,0.00012990429,0.00017407352,0.00024011816,0.0024782375,0.0123866685],"genre_scores_gemma":[0.92213583,0.0011519614,0.07539402,0.00015196965,0.00009865635,0.00005144287,0.000113587106,0.00013386167,0.00076870475],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98981655,0.004996632,0.00038934598,0.0005871852,0.00392152,0.00028864158],"domain_scores_gemma":[0.89125115,0.091361634,0.0039415,0.006717937,0.006354789,0.0003730584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006543355,0.0011200513,0.0007380307,0.00580374,0.0007779919,0.002279223,0.00087680964,0.000882963,0.0014516666],"category_scores_gemma":[0.042260434,0.00047726117,0.0007592982,0.0029583767,0.0034785962,0.0049676676,0.0007291718,0.0009961132,0.00054153294],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083894783,0.00035182276,0.078422844,0.0004414164,0.0003010699,0.00045129014,0.0006417152,0.17200801,0.019094637,0.024888566,0.0015298467,0.7010298],"study_design_scores_gemma":[0.0000784717,0.0016310149,0.055131998,0.00038462356,0.00039311356,0.0014348156,0.0005844062,0.7966369,0.037855394,0.09934291,0.0061585098,0.00036782015],"about_ca_topic_score_codex":0.002920281,"about_ca_topic_score_gemma":0.0025826269,"teacher_disagreement_score":0.006543355,"about_ca_system_score_codex":0.001047457,"about_ca_system_score_gemma":0.0011623607,"threshold_uncertainty_score":0.034604967},"labels":[],"label_agreement":null},{"id":"W2107915643","doi":"10.1109/tse.2013.20","title":"Whitening SOA Testing via Event Exposure","year":2013,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Service (business); Event (particle physics); Service provider; Information leakage; Implementation; Reliability engineering; Test (biology); Leakage (economics); Embedded system; Real-time computing; Computer security; Software engineering; Engineering","score_opus":0.01368091417226672,"score_gpt":0.2104804074451601,"score_spread":0.1967994932728934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2107915643","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08626238,0.00019500144,0.89985013,0.00024036903,0.000039925726,0.00019239962,0.00006953607,0.009721981,0.003428298],"genre_scores_gemma":[0.77416193,0.00016797896,0.22216786,0.00035535687,0.000044684242,0.00020732328,0.00020884833,0.0008065208,0.0018795562],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942543,0.0021491123,0.0003566314,0.00081496063,0.001995956,0.0004290439],"domain_scores_gemma":[0.9802868,0.011786061,0.0014133053,0.0050396915,0.0011713058,0.00030290938],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031676528,0.0012379392,0.0008654188,0.0018035088,0.00043401754,0.0018685643,0.0015142224,0.0011539634,0.0018118058],"category_scores_gemma":[0.013049924,0.00063532597,0.0012116618,0.0008638807,0.0016651921,0.003044821,0.002651686,0.0014984973,0.00042962152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013682559,0.0008015516,0.011475983,0.00039989257,0.00018795712,0.0018287902,0.001512399,0.1682404,0.16533017,0.039008487,0.002562254,0.60728395],"study_design_scores_gemma":[0.00012635517,0.0005200939,0.002058283,0.00010582913,0.00014099755,0.000802663,0.00013993178,0.6962402,0.24419622,0.04694556,0.008646197,0.0000777066],"about_ca_topic_score_codex":0.00089781696,"about_ca_topic_score_gemma":0.0007814119,"teacher_disagreement_score":0.0031676528,"about_ca_system_score_codex":0.0007307982,"about_ca_system_score_gemma":0.0009861299,"threshold_uncertainty_score":0.016752362},"labels":[],"label_agreement":null},{"id":"W2110431240","doi":"10.1109/tse.2002.1019484","title":"Assessing the applicability of fault-proneness models across object-oriented software projects","year":2002,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":351,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Universidade Federal do Rio Grande do Sul; National Science Foundation","keywords":"Computer science; Software metric; Software system; Software; Data mining; A priori and a posteriori; Multivariate adaptive regression splines; Software development; Java; Fault (geology); Software fault tolerance; Software engineering; Machine learning; Software quality; Regression analysis; Programming language","score_opus":0.04020972332741948,"score_gpt":0.2891038020238414,"score_spread":0.2488940786964219,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2110431240","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9522756,0.00025857575,0.04520004,0.0002831032,0.000011894052,0.00011058561,0.00036464032,0.00019108976,0.0013044978],"genre_scores_gemma":[0.98904115,0.00008385528,0.010141595,0.00002398865,0.000010173352,0.00007600174,0.00043562838,0.000027696244,0.00016005228],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.989421,0.0059182155,0.0006915259,0.0014149982,0.002021598,0.00053267594],"domain_scores_gemma":[0.84448105,0.1212119,0.015718376,0.010952653,0.0062187854,0.0014171948],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026272558,0.0011534463,0.00091653765,0.0054267147,0.0005754826,0.0016504949,0.002134443,0.0018882575,0.0008983266],"category_scores_gemma":[0.10249006,0.0005668817,0.0017470713,0.0034022925,0.0011576964,0.003849548,0.002505996,0.0018406687,0.00024460215],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075951003,0.00085127715,0.41136283,0.00017987624,0.00087060314,0.00017501798,0.0013263939,0.51517606,0.0014051794,0.0045370953,0.0005568232,0.06279925],"study_design_scores_gemma":[0.00003400454,0.0014241135,0.15601353,0.00004993236,0.00012369768,0.00011929553,0.000786612,0.8304014,0.00082289823,0.009556324,0.00060949,0.000058628357],"about_ca_topic_score_codex":0.0059622442,"about_ca_topic_score_gemma":0.004746999,"teacher_disagreement_score":0.026272558,"about_ca_system_score_codex":0.0017533789,"about_ca_system_score_gemma":0.00080031465,"threshold_uncertainty_score":0.13894421},"labels":[],"label_agreement":null},{"id":"W2111555385","doi":"10.1109/tse.2010.70","title":"Solving the Class Responsibility Assignment Problem in Object-Oriented Analysis with Multi-Objective Genetic Algorithms","year":2010,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":128,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Heuristics; Class (philosophy); Genetic algorithm; Cohesion (chemistry); Domain (mathematical analysis); Context (archaeology); Class diagram; Object-oriented programming; Algorithm; Machine learning; Artificial intelligence; Programming language; Mathematics; Unified Modeling Language; Software","score_opus":0.009970145975078455,"score_gpt":0.24159022935009608,"score_spread":0.23162008337501763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111555385","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042622704,0.00018163334,0.9551,0.00018836801,0.000018242397,0.00006618105,0.000013570233,0.00017217145,0.001637144],"genre_scores_gemma":[0.36823687,0.00020813692,0.62985575,0.000112188,0.00002221478,0.00019985947,0.00004391691,0.00008325003,0.0012377573],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920744,0.00043675138,0.000030345074,0.00010041872,0.00015422805,0.00007080374],"domain_scores_gemma":[0.9982084,0.0013428895,0.00018722491,0.000059456648,0.00015051565,0.000051500065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002317964,0.0009845312,0.00085134694,0.0013040017,0.0008032504,0.0012034951,0.0009739937,0.0014586229,0.00094693084],"category_scores_gemma":[0.004436675,0.0005998877,0.0007397118,0.00089269894,0.0011040715,0.0008876045,0.00087495986,0.0009506788,0.0001380574],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019604222,0.00003502528,0.000584025,0.000042489086,0.000029581593,0.00004022149,0.00008733598,0.96635544,0.00065110705,0.0062113344,0.00021386963,0.025729919],"study_design_scores_gemma":[0.000011942895,0.00001733558,0.000096227115,0.000009888722,0.000012356176,0.000014109717,0.000030335508,0.9922908,0.00046882793,0.0067099305,0.00033346887,0.000004800983],"about_ca_topic_score_codex":0.0059367674,"about_ca_topic_score_gemma":0.0056967414,"teacher_disagreement_score":0.0059367674,"about_ca_system_score_codex":0.0012263235,"about_ca_system_score_gemma":0.0018508389,"threshold_uncertainty_score":0.0122587085},"labels":[],"label_agreement":null},{"id":"W2112717272","doi":"10.1109/32.935852","title":"Foundations of the trace assertion method of module interface specification","year":2001,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Logic, programming, and type systems","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Assertion; TRACE (psycholinguistics); Computer science; Programming language; Notation; Interface (matter); Formal specification; Operating system; Arithmetic; Mathematics","score_opus":0.02656165542666112,"score_gpt":0.2678047199802575,"score_spread":0.2412430645535964,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2112717272","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000698117,0.000096375574,0.9950831,0.00017987101,0.000047375575,0.000065671375,0.000059564984,0.00090829085,0.0028616793],"genre_scores_gemma":[0.056815837,0.0005241807,0.9335195,0.0003088602,0.00026878883,0.0006606579,0.00042888822,0.0008551415,0.0066181505],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99156487,0.0025988698,0.0008035294,0.0012486781,0.0032001857,0.00058392424],"domain_scores_gemma":[0.9889101,0.005512108,0.00063380745,0.0024744908,0.002194954,0.0002745927],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009396584,0.0012861204,0.001010946,0.0027778957,0.0013728504,0.004760027,0.004659429,0.002754523,0.01044586],"category_scores_gemma":[0.022210915,0.0016343473,0.0027472824,0.0022378785,0.005858169,0.008450272,0.0037256428,0.0048908796,0.004162499],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004239358,0.00005104238,0.00025869874,0.00013977967,0.000024065344,0.0001572999,0.00056427374,0.0052128937,0.0014524469,0.9427233,0.0020306192,0.04734318],"study_design_scores_gemma":[0.00008605382,0.00009152505,0.0001489461,0.00018560258,0.0000475818,0.00031768504,0.000097242075,0.095100574,0.0074594575,0.82792693,0.06844264,0.00009576415],"about_ca_topic_score_codex":0.0033817384,"about_ca_topic_score_gemma":0.0017326361,"teacher_disagreement_score":0.01044586,"about_ca_system_score_codex":0.0021297734,"about_ca_system_score_gemma":0.0044238856,"threshold_uncertainty_score":0.04969448},"labels":[],"label_agreement":null},{"id":"W2112792171","doi":"10.1109/tse.2005.88","title":"Automatic inclusion of middleware performance attributes into architectural UML software models","year":2005,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Vlaamse regering; University of Ottawa","keywords":"Computer science; Middleware (distributed applications); Unified Modeling Language; Overhead (engineering); Message oriented middleware; Model transformation; Distributed computing; Software architecture; Programming paradigm; Software engineering; Software; Programming language; Artificial intelligence","score_opus":0.01152522820902501,"score_gpt":0.21113453026997295,"score_spread":0.19960930206094793,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2112792171","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013866978,0.00005127392,0.96425015,0.00011600378,0.000035496865,0.00016969156,0.00054815,0.019077357,0.0018848974],"genre_scores_gemma":[0.19190095,0.0002175541,0.7974753,0.000060404407,0.000034194167,0.00033146553,0.003190474,0.0042689964,0.0025207156],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970312,0.00083994033,0.0003155236,0.0003094211,0.0013719372,0.00013192274],"domain_scores_gemma":[0.9934356,0.0030030538,0.00069476134,0.0014239035,0.0013497501,0.00009297045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034824528,0.0013917185,0.00076717796,0.002383025,0.00066068576,0.0026934864,0.0014486746,0.0009882264,0.0027838885],"category_scores_gemma":[0.014613803,0.0015077228,0.0018854756,0.0011028327,0.0006270685,0.0028632625,0.0018312093,0.0018736573,0.0015047078],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064308936,0.0005182174,0.009887348,0.0007804406,0.00019217373,0.00096631265,0.004254757,0.40581748,0.053780857,0.09243218,0.0106684575,0.42005867],"study_design_scores_gemma":[0.00004238917,0.0000616306,0.00064585387,0.000093747454,0.000091052934,0.0001513611,0.00016016541,0.91697985,0.034186613,0.020278351,0.027250446,0.000058456557],"about_ca_topic_score_codex":0.003277126,"about_ca_topic_score_gemma":0.004153318,"teacher_disagreement_score":0.0034824528,"about_ca_system_score_codex":0.0014959944,"about_ca_system_score_gemma":0.0016859641,"threshold_uncertainty_score":0.01841718},"labels":[],"label_agreement":null},{"id":"W2116117396","doi":"10.1109/tse.2003.1237171","title":"Temporal logic query checking: a tool for model exploration","year":2003,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Temporal logic; Model checking; Kripke structure; Theoretical computer science; Query optimization; Computation tree logic; Linear temporal logic; Interval temporal logic; Query language; Set (abstract data type); Programming language; Information retrieval","score_opus":0.05622366089218801,"score_gpt":0.2785760117045132,"score_spread":0.22235235081232518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2116117396","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019504137,0.000090499176,0.9902238,0.0002543266,0.000027995824,0.00009045768,0.00014159457,0.0064103277,0.0008105014],"genre_scores_gemma":[0.111102335,0.00027037438,0.88480014,0.0003835187,0.00007700578,0.00042226544,0.0005573101,0.0011616015,0.0012253541],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9913693,0.0033007762,0.0007333376,0.001294448,0.0027967426,0.00050541287],"domain_scores_gemma":[0.9650836,0.025896141,0.0016470983,0.005233979,0.001718241,0.00042096205],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008106193,0.0016411961,0.0014480802,0.0029913383,0.0012587144,0.003432382,0.0038457264,0.0017631693,0.0052495827],"category_scores_gemma":[0.03484001,0.0016230489,0.003377285,0.0024195588,0.005528235,0.009750882,0.0064204936,0.0052054776,0.00095617457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006734086,0.00026774761,0.002942641,0.0010129717,0.00029009895,0.0011524614,0.001755689,0.08064788,0.017916013,0.61633635,0.015727269,0.2612774],"study_design_scores_gemma":[0.00015702358,0.00014912018,0.000233536,0.00019911354,0.00011785752,0.00054784876,0.00019460777,0.5701344,0.027414924,0.3762074,0.02454856,0.00009561584],"about_ca_topic_score_codex":0.0034903267,"about_ca_topic_score_gemma":0.0030481971,"teacher_disagreement_score":0.008106193,"about_ca_system_score_codex":0.0017364366,"about_ca_system_score_gemma":0.0028949098,"threshold_uncertainty_score":0.042870104},"labels":[],"label_agreement":null},{"id":"W2118604270","doi":"10.1109/tse.2012.28","title":"The Effects of Test-Driven Development on External Quality and Productivity: A Meta-Analysis","year":2013,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":125,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Productivity; Moderation; Computer science; Meta-analysis; Quality (philosophy); Task (project management); Quality management; Statistics; Operations management; Mathematics; Engineering; Economics","score_opus":0.03296412813828898,"score_gpt":0.27103694364331116,"score_spread":0.23807281550502218,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118604270","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019800285,0.9763016,0.0018154972,0.00026224885,0.00024478597,0.00045036193,0.00065277977,0.000055661727,0.00041671557],"genre_scores_gemma":[0.44631258,0.54264915,0.0062356815,0.0008380781,0.00030259232,0.0017728641,0.0012973384,0.00010354056,0.00048817086],"study_design_codex":"meta_analysis","study_design_gemma":"meta_analysis","domain_scores_codex":[0.98186284,0.009493257,0.004717534,0.001571752,0.0020023226,0.00035228883],"domain_scores_gemma":[0.94297147,0.0462486,0.005831053,0.0018941294,0.002543194,0.00051157904],"candidate_categories":["metaresearch","metaepi_broad"],"consensus_categories":[],"category_scores_codex":[0.025042087,0.0029373362,0.014091789,0.0074247764,0.0007245502,0.0038367494,0.0022295106,0.0019016683,0.0021358645],"category_scores_gemma":[0.06032579,0.0015542982,0.047083456,0.006997239,0.0008938088,0.0018648756,0.00178725,0.0022395328,0.00021735948],"study_design_candidate":"meta_analysis","study_design_consensus":"meta_analysis","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002152672,0.000038809747,0.0046465355,0.13113195,0.8508765,0.000107422245,0.00009128379,0.00040926068,0.00028661004,0.00009181698,0.00024871054,0.00991849],"study_design_scores_gemma":[0.00047421287,0.00024292644,0.0034468102,0.008394125,0.98627937,0.000071658214,0.00003344129,0.00010144776,0.00022172074,0.00011068706,0.0006091558,0.000014430292],"about_ca_topic_score_codex":0.004917013,"about_ca_topic_score_gemma":0.009458418,"teacher_disagreement_score":0.9859082,"about_ca_system_score_codex":0.0031145776,"about_ca_system_score_gemma":0.0027136186,"threshold_uncertainty_score":0.13243681},"labels":[],"label_agreement":null},{"id":"W2120738100","doi":"10.1109/tse.2006.102","title":"Empirical Analysis of Object-Oriented Design Metrics for Predicting High and Low Severity Faults","year":2006,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":349,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Larner College of Medicine, University of Vermont; University of Alberta","keywords":"Computer science; Fault (geology); Data mining; Software fault tolerance; Empirical research; Object-oriented programming; Machine learning; Reliability engineering; Logistic regression; Artificial intelligence; Fault tolerance; Distributed computing; Engineering; Statistics; Mathematics; Programming language","score_opus":0.018279048185542823,"score_gpt":0.260653268209901,"score_spread":0.24237422002435816,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120738100","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9877809,0.0003834303,0.010417827,0.00013592624,0.000007811094,0.000026371798,0.0006326763,0.0001194547,0.00049562607],"genre_scores_gemma":[0.99547064,0.00008215173,0.003403874,0.000011003432,0.0000075250014,0.000017168843,0.00092830666,0.000013093194,0.00006630466],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9964851,0.0016256283,0.00033331005,0.00033119216,0.0010808414,0.00014388216],"domain_scores_gemma":[0.89432627,0.08083223,0.013834485,0.0042202175,0.005693038,0.0010937087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068428507,0.0008595249,0.00041916792,0.004498734,0.00020297998,0.0006259704,0.00043189412,0.0006175622,0.0004445636],"category_scores_gemma":[0.060883675,0.00016787206,0.00039480889,0.0029093192,0.00044463226,0.0011643238,0.00046845397,0.0006937869,0.00024144241],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012037052,0.00014060948,0.94749177,0.000052971845,0.00013375249,0.000053105276,0.000101357306,0.022021312,0.00063269626,0.00020912803,0.0004159277,0.028627088],"study_design_scores_gemma":[0.000035408026,0.0004128474,0.71041214,0.000038068138,0.00007028013,0.00025630955,0.00019231495,0.2830052,0.0027668553,0.0018722058,0.00090434076,0.000034007106],"about_ca_topic_score_codex":0.0019360883,"about_ca_topic_score_gemma":0.0020999638,"teacher_disagreement_score":0.0068428507,"about_ca_system_score_codex":0.000403286,"about_ca_system_score_gemma":0.00039549242,"threshold_uncertainty_score":0.0361889},"labels":[],"label_agreement":null},{"id":"W2120755390","doi":"10.1109/tse.2014.2387172","title":"Extracting Development Tasks to Navigate Software Documentation","year":2014,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":80,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Documentation; Software documentation; Computer science; Internal documentation; Software engineering; Software development; Application programming interface; Technical documentation; Software; Task (project management); World Wide Web; Software construction; Programming language; Systems engineering; Engineering","score_opus":0.013513800978778162,"score_gpt":0.25689729459795757,"score_spread":0.2433834936191794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120755390","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40441975,0.0028490087,0.53192824,0.0017740333,0.0002091743,0.0026305034,0.017481927,0.025296109,0.013411326],"genre_scores_gemma":[0.253318,0.0012263015,0.714933,0.00025362446,0.000057132314,0.0011740024,0.022569412,0.0019078271,0.0045607863],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967,0.0010436119,0.00064453087,0.0004638204,0.0009253863,0.00022267828],"domain_scores_gemma":[0.9675826,0.020962434,0.0028368304,0.0024085983,0.005583963,0.00062560197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035798384,0.0017197339,0.0009992875,0.011277053,0.0012413196,0.002248227,0.0011382418,0.0011491199,0.0023018003],"category_scores_gemma":[0.037713587,0.00077417534,0.0009044351,0.0050535686,0.00037208622,0.0031590192,0.0021989872,0.0010306053,0.002167296],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064611866,0.0005117638,0.0376596,0.005005767,0.000099633005,0.0012736638,0.015810113,0.004753964,0.033847585,0.00577627,0.048684627,0.845931],"study_design_scores_gemma":[0.0005365692,0.0016878757,0.102924235,0.00500647,0.0005136765,0.0043935087,0.03144958,0.22040917,0.13178794,0.03643457,0.46407583,0.00078055327],"about_ca_topic_score_codex":0.0059148134,"about_ca_topic_score_gemma":0.012246673,"teacher_disagreement_score":0.011277053,"about_ca_system_score_codex":0.00095525576,"about_ca_system_score_gemma":0.0037601423,"threshold_uncertainty_score":0.018932223},"labels":[],"label_agreement":null},{"id":"W2122581326","doi":"10.1109/tse.2008.36","title":"Do Crosscutting Concerns Cause Defects?","year":2008,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":238,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of Calgary","funders":"","keywords":"Computer science; Harm; Process (computing); Code (set theory); Measure (data warehouse); Source code; Degree (music); Software engineering; Risk analysis (engineering); Reliability engineering; Data mining; Programming language; Law; Engineering","score_opus":0.03323913553247372,"score_gpt":0.26905700605010585,"score_spread":0.23581787051763214,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122581326","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9711368,0.0018717622,0.017798536,0.0022059586,0.00005769168,0.00007994906,0.0001806449,0.00018295707,0.006485745],"genre_scores_gemma":[0.99630094,0.00038871358,0.0025288034,0.00030754905,0.000024146399,0.00002635846,0.000084087646,0.000030138272,0.0003092915],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9887832,0.0032332009,0.00068864325,0.0019640469,0.0048028054,0.000528164],"domain_scores_gemma":[0.75856733,0.17880648,0.040913995,0.008663506,0.01160007,0.0014485464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008900899,0.00074954325,0.000618793,0.00284359,0.0006285419,0.0014615393,0.0010076741,0.002403595,0.0026320193],"category_scores_gemma":[0.10078901,0.0006390598,0.0007191138,0.002047856,0.0022937777,0.0041517345,0.0012760164,0.0013692583,0.00024735762],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004562408,0.00079616264,0.81277245,0.0013844214,0.0005794577,0.0016029058,0.005722561,0.002528745,0.025139961,0.012549148,0.0012016356,0.13526647],"study_design_scores_gemma":[0.000104294464,0.0014268291,0.88857555,0.000447707,0.000660256,0.0051280996,0.007437159,0.009449279,0.04469799,0.034781355,0.007200688,0.00009072414],"about_ca_topic_score_codex":0.0012998249,"about_ca_topic_score_gemma":0.0019906075,"teacher_disagreement_score":0.008900899,"about_ca_system_score_codex":0.00091396115,"about_ca_system_score_gemma":0.0008329307,"threshold_uncertainty_score":0.047073007},"labels":[],"label_agreement":null},{"id":"W2128089903","doi":"10.1109/32.917521","title":"Design, construction, and application of a generic visual language generation environment","year":2001,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre For Cold Ocean Resources Engineering","funders":"","keywords":"Computer science; Programming language; Compiler; Visual programming language; Visual language; Grammar; Parsing; Artificial intelligence","score_opus":0.010869366349029446,"score_gpt":0.21358298539060863,"score_spread":0.2027136190415792,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128089903","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005592177,0.00004734568,0.98061484,0.000059117232,0.00002160931,0.00019943144,0.00007097294,0.011411779,0.001982757],"genre_scores_gemma":[0.05822285,0.00012934378,0.93591636,0.00011752685,0.000018264427,0.00030129138,0.0004987415,0.0023950248,0.0024006388],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99868923,0.00032297068,0.00012143634,0.00042185126,0.00033780801,0.000106754786],"domain_scores_gemma":[0.9987068,0.0004076083,0.0001145025,0.00041202916,0.00024757907,0.00011133433],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022185706,0.00076868327,0.0005359034,0.0010079555,0.00041866477,0.0018494767,0.0021482164,0.0011554664,0.0029453041],"category_scores_gemma":[0.0035897903,0.00091953704,0.000905459,0.0004917504,0.0011093376,0.0020722186,0.0019901206,0.0014140205,0.0013203917],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005628713,0.000461958,0.0035181788,0.001080245,0.00011926353,0.0022210681,0.002717233,0.04380239,0.18922073,0.130424,0.016191797,0.6096803],"study_design_scores_gemma":[0.00050111307,0.0007899367,0.0026477273,0.00029230752,0.00018179376,0.0035560662,0.00042450218,0.4313758,0.2757124,0.036581278,0.24768302,0.00025413054],"about_ca_topic_score_codex":0.0004907535,"about_ca_topic_score_gemma":0.0003614703,"teacher_disagreement_score":0.0029453041,"about_ca_system_score_codex":0.00057818514,"about_ca_system_score_gemma":0.0008311602,"threshold_uncertainty_score":0.011733055},"labels":[],"label_agreement":null},{"id":"W2129603121","doi":"10.1109/tse.2005.15","title":"Toward formalizing domain modeling semantics in language syntax","year":2005,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Service-Oriented Architecture and Web Services","field":"Computer Science","cited_by":80,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Metamodeling; Unified Modeling Language; Business domain; Modeling language; Domain model; Domain engineering; Domain (mathematical analysis); Programming language; Domain analysis; Object Constraint Language; Semantics (computer science); Software engineering; Syntax; Applications of UML; Natural language processing; Business rule; Software development; Domain knowledge; Business process; Component-based software engineering; Work in process","score_opus":0.009961354837146816,"score_gpt":0.21386160839263096,"score_spread":0.20390025355548413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129603121","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013204797,0.00016626161,0.9963372,0.00040604844,0.000045481665,0.0000563911,0.00005758069,0.00029193712,0.0013186128],"genre_scores_gemma":[0.045022033,0.00081987213,0.95098907,0.00047284272,0.0001725454,0.0005170513,0.00044651024,0.00040539724,0.0011546006],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9877814,0.006312034,0.0018577967,0.0013417538,0.0020834005,0.00062362035],"domain_scores_gemma":[0.98087054,0.008998638,0.0021330307,0.0046196263,0.0028797332,0.00049841264],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019213296,0.0023174721,0.0016775515,0.0037009162,0.0026401142,0.011575186,0.0045221443,0.0035397424,0.0027372038],"category_scores_gemma":[0.020424556,0.0023115547,0.004884715,0.0033039968,0.008554265,0.022909377,0.007681528,0.011285074,0.0013561306],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012681791,0.000030901498,0.0002228639,0.000106928215,0.000024478664,0.00008945304,0.001110456,0.008316118,0.0011957681,0.9789013,0.0006332937,0.009355695],"study_design_scores_gemma":[0.000031173226,0.000038964045,0.000096170676,0.00035042735,0.000079769896,0.00021082835,0.00053548906,0.07076385,0.0036006465,0.87495387,0.0492783,0.00006052604],"about_ca_topic_score_codex":0.0039736945,"about_ca_topic_score_gemma":0.00422558,"teacher_disagreement_score":0.019213296,"about_ca_system_score_codex":0.0027105105,"about_ca_system_score_gemma":0.007733679,"threshold_uncertainty_score":0.1016109},"labels":[],"label_agreement":null},{"id":"W2130400275","doi":"10.1109/32.888631","title":"A learning agent that assists the browsing of software libraries","year":2000,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Machine Learning and Algorithms","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Software; Similarity (geometry); World Wide Web; Information retrieval; Human–computer interaction; Artificial intelligence; Programming language","score_opus":0.010120749815697744,"score_gpt":0.20454357418451105,"score_spread":0.1944228243688133,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2130400275","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23626418,0.00046707742,0.72430396,0.00064189994,0.000088987974,0.0005676953,0.00025561033,0.022611918,0.014798672],"genre_scores_gemma":[0.458794,0.00025316665,0.52678216,0.0003289551,0.000063950996,0.00018173002,0.00036157013,0.00026443505,0.012970008],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99959224,0.00011270141,0.000027542339,0.00010439024,0.00012717559,0.00003597063],"domain_scores_gemma":[0.9969406,0.0016182411,0.0003427331,0.00037529244,0.0004200081,0.00030313202],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010084122,0.0005725755,0.00068879285,0.00066702295,0.0004716116,0.00081165205,0.0014488085,0.0013606683,0.0041021165],"category_scores_gemma":[0.005849833,0.00033755036,0.00023604915,0.0005949257,0.0004082149,0.0020035545,0.0007686495,0.0011529614,0.0025506138],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009559343,0.0026120546,0.013648884,0.0003660117,0.00007074352,0.0005000638,0.0007176533,0.016412184,0.040300686,0.003513203,0.011130496,0.9097721],"study_design_scores_gemma":[0.0003853738,0.0015927067,0.005621688,0.00007917842,0.00020098283,0.0017133942,0.00039212426,0.8666246,0.07353491,0.0058398214,0.043905806,0.00010944382],"about_ca_topic_score_codex":0.0016756152,"about_ca_topic_score_gemma":0.003408259,"teacher_disagreement_score":0.0041021165,"about_ca_system_score_codex":0.00028062484,"about_ca_system_score_gemma":0.0009172698,"threshold_uncertainty_score":0.013722956},"labels":[],"label_agreement":null},{"id":"W2131878172","doi":"10.1109/tse.2003.1237169","title":"Template semantics for model-based notations","year":2003,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":58,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"University of Waterloo; University of Minnesota","keywords":"Computer science; Notation; Semantics (computer science); Programming language; Operational semantics; Structuring; Denotational semantics; Action semantics; Variety (cybernetics); Well-founded semantics; Theoretical computer science; Artificial intelligence; Mathematics","score_opus":0.028663676038960133,"score_gpt":0.2666123737194766,"score_spread":0.23794869768051646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2131878172","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011268362,0.00015217773,0.99330974,0.00031938197,0.000103116006,0.000112455105,0.00014322248,0.0009517887,0.0037811946],"genre_scores_gemma":[0.08119834,0.0007390454,0.90972275,0.00061174604,0.00025556562,0.0012310359,0.0010202031,0.00079800346,0.004423312],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9887117,0.0042177862,0.002247497,0.0011463922,0.0030113065,0.00066522305],"domain_scores_gemma":[0.99036086,0.0037600875,0.00096061715,0.0025227128,0.0019647423,0.00043097243],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009548809,0.0017764601,0.0012973968,0.0023498836,0.0016607163,0.008070885,0.0039779884,0.0030161778,0.0053002955],"category_scores_gemma":[0.014169874,0.0013779537,0.0035665128,0.0025219554,0.0049312324,0.0124244355,0.0037097933,0.004648354,0.0025219123],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012989414,0.0000128940965,0.000052857424,0.000078235105,0.000010864879,0.00008266545,0.0003967463,0.0031874992,0.0006995643,0.9885754,0.000998746,0.005891635],"study_design_scores_gemma":[0.000036524027,0.000047387104,0.000028255849,0.000121289704,0.00003380718,0.00021451976,0.00015053991,0.032214113,0.0034959032,0.8856237,0.077994354,0.000039572344],"about_ca_topic_score_codex":0.0022632384,"about_ca_topic_score_gemma":0.0016982993,"teacher_disagreement_score":0.009548809,"about_ca_system_score_codex":0.002759572,"about_ca_system_score_gemma":0.0033008282,"threshold_uncertainty_score":0.05049956},"labels":[],"label_agreement":null},{"id":"W2136921053","doi":"10.1109/tse.2010.58","title":"The Effects of Time Constraints on Test Case Prioritization: A Series of Controlled Experiments","year":2010,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":166,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Regression testing; Prioritization; Computer science; Context (archaeology); Risk-based testing; Reliability engineering; Process (computing); Software; Regression analysis; Data mining; Risk analysis (engineering); Machine learning; Software development; Engineering; Management science; Software construction","score_opus":0.005484845015205005,"score_gpt":0.2193811498363502,"score_spread":0.21389630482114522,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136921053","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9740252,0.00038227,0.011603265,0.00020656876,0.0003282995,0.010084174,0.0005800266,0.00029924072,0.0024911305],"genre_scores_gemma":[0.872146,0.00060133537,0.065419614,0.0012052848,0.00033703307,0.05410937,0.00113904,0.0002570871,0.0047852406],"study_design_codex":"nonrandomized_trial","study_design_gemma":"nonrandomized_trial","domain_scores_codex":[0.9861057,0.0058070086,0.0018751144,0.0027173096,0.0022042894,0.0012906112],"domain_scores_gemma":[0.78933585,0.17330246,0.0181555,0.008236971,0.00777192,0.0031972928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014764209,0.0034786686,0.0017620694,0.0012345327,0.0013124457,0.0021484618,0.004320335,0.003055643,0.0058646575],"category_scores_gemma":[0.065497786,0.0013504918,0.0016918764,0.0013101477,0.0027290937,0.00241166,0.0017960834,0.0048815105,0.0006025335],"study_design_candidate":"nonrandomized_trial","study_design_consensus":"nonrandomized_trial","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.18169585,0.33123028,0.016931133,0.0049622147,0.0022530088,0.0011127874,0.008107449,0.045927897,0.2677897,0.006374319,0.0049967812,0.12861867],"study_design_scores_gemma":[0.06333205,0.6317912,0.046115436,0.0005238389,0.0030226633,0.0003337327,0.0019960832,0.055294324,0.16689493,0.013361185,0.016385296,0.0009493094],"about_ca_topic_score_codex":0.0024355275,"about_ca_topic_score_gemma":0.0028702493,"teacher_disagreement_score":0.014764209,"about_ca_system_score_codex":0.0025654656,"about_ca_system_score_gemma":0.002879321,"threshold_uncertainty_score":0.07808149},"labels":[],"label_agreement":null},{"id":"W2138201395","doi":"10.1109/tse.2003.1214324","title":"An investigation of graph-based class integration test order strategies","year":2003,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":116,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada; Goddard Space Flight Center","keywords":"Computer science; Dependency (UML); Dependency graph; Inheritance (genetic algorithm); Class (philosophy); Theoretical computer science; Graph; Context (archaeology); Artificial intelligence","score_opus":0.01487302891763947,"score_gpt":0.2431641625133235,"score_spread":0.22829113359568404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2138201395","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33523017,0.00090836495,0.6377694,0.00095376396,0.000036265603,0.0008063555,0.0001512122,0.0008616716,0.023282835],"genre_scores_gemma":[0.81799,0.00043343235,0.17971374,0.00010680978,0.000009754321,0.00017573885,0.00017209306,0.000095265954,0.001303199],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99653155,0.0016842732,0.00016407335,0.00026980243,0.0011468196,0.00020346597],"domain_scores_gemma":[0.97082055,0.023430567,0.0015584956,0.001302635,0.0025752753,0.0003124381],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004078192,0.00072744535,0.00051055785,0.0032901824,0.00049471005,0.0018573727,0.0013156252,0.0009544939,0.0031989156],"category_scores_gemma":[0.02720609,0.0003657544,0.0004998556,0.0018948094,0.0013624071,0.0033377542,0.0007291127,0.00077658525,0.0003514388],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004125103,0.00085722894,0.017945087,0.00058026524,0.00010242437,0.0005240155,0.0020510873,0.24283512,0.013484426,0.2530182,0.0020188799,0.46617073],"study_design_scores_gemma":[0.00015173861,0.0008144247,0.0050106873,0.00012069332,0.0001247142,0.00045582379,0.0010220661,0.88074565,0.013110685,0.09126527,0.0071164225,0.0000619403],"about_ca_topic_score_codex":0.0054241186,"about_ca_topic_score_gemma":0.004907154,"teacher_disagreement_score":0.0054241186,"about_ca_system_score_codex":0.0022090296,"about_ca_system_score_gemma":0.0021081034,"threshold_uncertainty_score":0.021567822},"labels":[],"label_agreement":null},{"id":"W2142033303","doi":"10.1109/32.888627","title":"Management of performance requirements for information systems","year":2000,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Non-functional requirement; System requirements; Software engineering; Requirements analysis; Requirements management; Process (computing); Domain (mathematical analysis); Requirements engineering; Systems engineering; Information system; Functional requirement; Variety (cybernetics); Software system; Software; Engineering; Software construction; Artificial intelligence","score_opus":0.010316117302928352,"score_gpt":0.21488042325859766,"score_spread":0.2045643059556693,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142033303","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04641234,0.0029913771,0.8140489,0.017481735,0.00024749036,0.0011333487,0.00048109214,0.0027346085,0.114469014],"genre_scores_gemma":[0.5304013,0.003697476,0.45113492,0.0014131824,0.0006148483,0.0013761489,0.0016325491,0.0008770515,0.008852554],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.96903414,0.0084925005,0.0029157647,0.0009968058,0.016466666,0.0020941526],"domain_scores_gemma":[0.9545152,0.022245787,0.0052835755,0.005819779,0.010866206,0.0012694465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018985514,0.0010842496,0.00080960814,0.0032007864,0.0027365908,0.01087092,0.0023343996,0.0028044798,0.0021414733],"category_scores_gemma":[0.057806052,0.0007586227,0.0008697001,0.0023057417,0.0023711121,0.010769551,0.0036712897,0.0032988868,0.0010978858],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006382556,0.00021106903,0.003329256,0.0010918267,0.000067659,0.00075283257,0.003188344,0.052951146,0.011694136,0.6598098,0.018299825,0.24854024],"study_design_scores_gemma":[0.00006113529,0.00039985537,0.00763335,0.0014721779,0.00010116051,0.0015603994,0.003959671,0.15785179,0.014550466,0.52815133,0.28392485,0.000333785],"about_ca_topic_score_codex":0.004614743,"about_ca_topic_score_gemma":0.0032587517,"teacher_disagreement_score":0.018985514,"about_ca_system_score_codex":0.0047526574,"about_ca_system_score_gemma":0.008233088,"threshold_uncertainty_score":0.10040617},"labels":[],"label_agreement":null},{"id":"W2145386371","doi":"10.1109/tse.2011.29","title":"Does Socio-Technical Congruence Have an Effect on Software Build Success? A Study of Coordination in a Software Project","year":2011,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Congruence (geometry); IBM; Computer science; Software; Knowledge management; Psychology; Programming language; Social psychology","score_opus":0.018081194288832585,"score_gpt":0.26614485705394325,"score_spread":0.24806366276511066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145386371","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99590874,0.000076581375,0.0011565771,0.00021915954,0.0000061811747,0.00001277579,0.000016779724,0.0000059622616,0.0025973187],"genre_scores_gemma":[0.99963844,0.000022957454,0.00024946037,0.000010339676,0.0000065356026,0.00000763527,0.000008985921,0.0000036523109,0.000051913346],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.985696,0.008947472,0.0006708397,0.0009564271,0.0025639217,0.0011653951],"domain_scores_gemma":[0.79872376,0.12854296,0.04419227,0.008163289,0.006600568,0.013777115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011409967,0.00038363613,0.0005669987,0.0024319594,0.00158161,0.0027452125,0.0007594281,0.0008277841,0.0030173713],"category_scores_gemma":[0.10853904,0.00036716228,0.00059736497,0.0027136186,0.0037533126,0.003039664,0.0037760863,0.0014825776,0.0004044907],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021881956,0.00021841322,0.98157173,0.00003106213,0.00010654665,0.0001574626,0.0035866168,0.0009148996,0.00052425044,0.0019014864,0.0000950686,0.010673684],"study_design_scores_gemma":[0.000019911706,0.00035044222,0.9902884,0.000016780392,0.000043222648,0.00009006587,0.003526916,0.002404635,0.00026805594,0.0026717442,0.00030154613,0.000018456141],"about_ca_topic_score_codex":0.002334516,"about_ca_topic_score_gemma":0.002469397,"teacher_disagreement_score":0.011409967,"about_ca_system_score_codex":0.00094936363,"about_ca_system_score_gemma":0.0013702993,"threshold_uncertainty_score":0.06034237},"labels":[],"label_agreement":null},{"id":"W2145512597","doi":"10.1109/tse.2010.91","title":"Work Item Tagging: Communicating Concerns in Collaborative Software Development","year":2010,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Categorization; Software development; Knowledge management; Mechanism (biology); Work (physics); Domain (mathematical analysis); Software; Vocabulary; Empirical research; Open-source software development; World Wide Web; Software engineering; Data science; Human–computer interaction; Artificial intelligence","score_opus":0.018162273126256012,"score_gpt":0.26037683232553865,"score_spread":0.24221455919928264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145512597","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5092361,0.00039908593,0.47395706,0.0015655408,0.00016455137,0.00060682173,0.00009958325,0.0011382158,0.012832951],"genre_scores_gemma":[0.9094877,0.00012704397,0.08746209,0.0002278391,0.00006513938,0.00028009035,0.00015072565,0.00015477427,0.0020445476],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9395561,0.046860512,0.003241386,0.0031017326,0.0060857683,0.0011545164],"domain_scores_gemma":[0.7933839,0.15259016,0.021327868,0.021165388,0.008843099,0.00268963],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03132726,0.0011716936,0.0006533076,0.003176606,0.0047116783,0.006580211,0.0024758433,0.0031772205,0.0015950949],"category_scores_gemma":[0.13499509,0.001057364,0.00071278756,0.0023501664,0.004995032,0.011553073,0.008969507,0.0023543986,0.0007920318],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010186497,0.0006728197,0.14736083,0.0011050138,0.00017373284,0.0019869914,0.33620694,0.0032274423,0.031862386,0.032003377,0.003038152,0.44134364],"study_design_scores_gemma":[0.0003619476,0.0034492188,0.20812926,0.0018061654,0.00085651915,0.0088522965,0.23608063,0.09066894,0.0692078,0.18838382,0.19074383,0.0014595567],"about_ca_topic_score_codex":0.0023048664,"about_ca_topic_score_gemma":0.0019266667,"teacher_disagreement_score":0.03132726,"about_ca_system_score_codex":0.0020538666,"about_ca_system_score_gemma":0.0025652093,"threshold_uncertainty_score":0.16567636},"labels":[],"label_agreement":null},{"id":"W2147386665","doi":"10.1109/tse.2012.70","title":"A large-scale empirical study of just-in-time quality assurance","year":2012,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":699,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Polytechnique Montréal","funders":"","keywords":"Computer science; Software quality assurance; Quality assurance; Source lines of code; Quality (philosophy); Software quality; Code review; Scale (ratio); Software; Empirical research; Software quality analyst; Software bug; Data science; Software engineering; Software development; Operations management; Engineering","score_opus":0.03357803100855729,"score_gpt":0.3186450643686469,"score_spread":0.28506703336008965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2147386665","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9964929,0.00035352763,0.0012871032,0.0003263917,0.000011803026,0.00003983928,0.0002604744,0.000023607097,0.0012042498],"genre_scores_gemma":[0.99857664,0.00013886501,0.00052927283,0.00008824062,0.0000129419195,0.000032100153,0.000356343,0.000013978826,0.00025147386],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9864699,0.0070611006,0.00082634226,0.0020013894,0.003177015,0.00046418345],"domain_scores_gemma":[0.5754482,0.33543867,0.046446085,0.0183386,0.018926969,0.0054015033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018737236,0.0005504697,0.0005454523,0.0021346756,0.0007920336,0.0012109333,0.0017413202,0.0010712658,0.0023887372],"category_scores_gemma":[0.13169342,0.00044346103,0.00063359295,0.0028554518,0.0016246059,0.003665669,0.0011714441,0.0026083754,0.00078523334],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015156965,0.00089636527,0.9746695,0.0001372657,0.0002684458,0.00017970183,0.0021773754,0.0016357413,0.00023287194,0.00068156637,0.0023869884,0.016582625],"study_design_scores_gemma":[0.00003337729,0.00056070765,0.9794597,0.00009669846,0.00007268736,0.00028638434,0.0019236719,0.013597496,0.00038079298,0.00053733075,0.003017405,0.00003376645],"about_ca_topic_score_codex":0.0071114493,"about_ca_topic_score_gemma":0.006842744,"teacher_disagreement_score":0.018737236,"about_ca_system_score_codex":0.0011202066,"about_ca_system_score_gemma":0.0007029035,"threshold_uncertainty_score":0.0990932},"labels":[],"label_agreement":null},{"id":"W2150853641","doi":"10.1109/tse.2011.65","title":"WAM—The Weighted Average Method for Predicting the Performance of Systems with Bursts of Customer Sessions","year":2011,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Benchmark (surveying); Queueing theory; Workload; Markov chain; Sizing; Benchmarking; Machine learning","score_opus":0.013726318271697586,"score_gpt":0.22237533266721862,"score_spread":0.20864901439552103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150853641","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02040733,0.00024671247,0.97638834,0.00007608979,0.000040068517,0.00004140328,0.00021187737,0.0019602594,0.00062788365],"genre_scores_gemma":[0.6029699,0.00045522058,0.3936886,0.00011344836,0.00010230143,0.0002581153,0.00074648083,0.00034491782,0.001321001],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99914813,0.00029423778,0.0000481163,0.00014932248,0.0002977468,0.00006242415],"domain_scores_gemma":[0.9967224,0.0018905266,0.00049227045,0.00039969943,0.0004133971,0.00008179398],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017296828,0.001378419,0.0010526512,0.0018805938,0.00046701965,0.0008149962,0.001745974,0.000809857,0.0011537997],"category_scores_gemma":[0.008501486,0.00053383707,0.0008734504,0.0014512768,0.0003420152,0.0018848834,0.00069095654,0.0012252231,0.0004590223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011174678,0.00007781536,0.0072341305,0.00009172398,0.0001811233,0.00006178887,0.000074416406,0.84199494,0.004624907,0.005223131,0.0022829857,0.13804129],"study_design_scores_gemma":[0.0000020864666,0.000017217408,0.0002892485,0.0000046844966,0.0000053430563,0.000012805633,0.0000045480965,0.99702483,0.00051754515,0.0018344535,0.00028031238,0.0000069724747],"about_ca_topic_score_codex":0.009199752,"about_ca_topic_score_gemma":0.006545497,"teacher_disagreement_score":0.009199752,"about_ca_system_score_codex":0.0006256352,"about_ca_system_score_gemma":0.0011075663,"threshold_uncertainty_score":0.018292367},"labels":[],"label_agreement":null},{"id":"W2151187574","doi":"10.1109/tse.2004.1274044","title":"An empirical study of open-source and closed-source software products","year":2004,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":317,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"General Dynamics (Canada)","funders":"University of Calgary","keywords":"Computer science; Open source; Empirical research; Open source software; Metric (unit); Software; Modular design; Software development; Software engineering; Empirical evidence; Creativity; Data science; Knowledge management; Marketing; Programming language; Statistics; Business; Mathematics","score_opus":0.021155778114453634,"score_gpt":0.2799976163484172,"score_spread":0.25884183823396356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2151187574","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9970071,0.00013019824,0.0006009505,0.00012988462,0.0000036155973,0.00007195375,0.00007146574,0.00000399398,0.0019809732],"genre_scores_gemma":[0.9977642,0.00014292377,0.001073573,0.00006643695,0.000008953563,0.00013364665,0.00019446384,0.000008361372,0.00060743216],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98528314,0.0072427737,0.0010285338,0.0010824566,0.0045524905,0.00081066997],"domain_scores_gemma":[0.70379853,0.2218021,0.030653866,0.008945221,0.026679657,0.008120583],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.014948062,0.00030521504,0.00042461132,0.0028543647,0.0017409507,0.002465878,0.0010133577,0.00095495733,0.0025444962],"category_scores_gemma":[0.121075965,0.00040591398,0.00027624323,0.0042240433,0.002638696,0.005879536,0.0024521905,0.0018887529,0.0005163761],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005185794,0.0034670332,0.8606726,0.0004341794,0.000056295255,0.0009064153,0.072028264,0.0004119934,0.0016774698,0.00536625,0.0011458289,0.053315017],"study_design_scores_gemma":[0.000072869494,0.00214876,0.85531586,0.0003193225,0.000027813672,0.0009964769,0.122623704,0.0027139552,0.0013297494,0.0020980863,0.012282653,0.00007069169],"about_ca_topic_score_codex":0.0017830213,"about_ca_topic_score_gemma":0.0019524465,"teacher_disagreement_score":0.99898666,"about_ca_system_score_codex":0.0011367151,"about_ca_system_score_gemma":0.0013570796,"threshold_uncertainty_score":0.07905382},"labels":[],"label_agreement":null},{"id":"W2151877107","doi":"10.1109/tse.2004.69","title":"A cognitive-based mechanism for constructing software inspection teams","year":2004,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Software inspection; Mechanism (biology); Process (computing); Selection (genetic algorithm); Software; Cognition; Artificial intelligence; Software engineering; Software development; Software quality","score_opus":0.015596019644818788,"score_gpt":0.24884721502036167,"score_spread":0.23325119537554287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2151877107","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039371423,0.000091650785,0.93879735,0.0008729801,0.00008930394,0.00042158418,0.000037819005,0.0011351957,0.01918266],"genre_scores_gemma":[0.48908058,0.00007565448,0.50294495,0.00023079752,0.00007980851,0.00082699314,0.00009308821,0.00007363953,0.006594443],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9935916,0.0021915745,0.00041006625,0.0013222591,0.001941071,0.0005434114],"domain_scores_gemma":[0.98645455,0.004596057,0.0023674546,0.0027009936,0.0024229556,0.0014580963],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008816103,0.0009072998,0.0004892036,0.0037473408,0.0030147084,0.0045898077,0.0043542194,0.002488966,0.0070862086],"category_scores_gemma":[0.029339818,0.00070987706,0.0013852997,0.0015624013,0.0037532786,0.0047761244,0.00468692,0.0015773808,0.0014689539],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026974603,0.00070541847,0.009203213,0.00024583362,0.00023653563,0.00055333617,0.006779901,0.03747665,0.016881244,0.7016523,0.0051294523,0.22086638],"study_design_scores_gemma":[0.0003881864,0.0009899476,0.0058379094,0.00018770681,0.0002458149,0.00092889677,0.0020912339,0.34709403,0.013832598,0.5870764,0.041011747,0.00031542347],"about_ca_topic_score_codex":0.00231458,"about_ca_topic_score_gemma":0.0021352211,"teacher_disagreement_score":0.008816103,"about_ca_system_score_codex":0.0021903715,"about_ca_system_score_gemma":0.0034584042,"threshold_uncertainty_score":0.0466246},"labels":[],"label_agreement":null},{"id":"W2159484301","doi":"10.1109/tse.2008.74","title":"Enhanced Modeling and Solution of Layered Queueing Networks","year":2008,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":204,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University; IBM (Canada); Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Queueing theory; Distributed computing; Layered queueing network; Dependability; Control reconfiguration; Canonical form; Queue; Representation (politics); Theoretical computer science; Computer network","score_opus":0.012426582439109008,"score_gpt":0.20113252693447015,"score_spread":0.18870594449536113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159484301","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015621007,0.00024014009,0.9795989,0.00020382875,0.000048206384,0.00003403034,0.000085810476,0.00022001447,0.003948029],"genre_scores_gemma":[0.6048783,0.0008008456,0.3848329,0.0001282836,0.00010294979,0.0002591279,0.0002879832,0.00014920975,0.008560445],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99929416,0.00027623828,0.000032936565,0.000075834585,0.00023027476,0.00009059902],"domain_scores_gemma":[0.9989967,0.0004780177,0.00012671167,0.00008219622,0.00025264864,0.000063813815],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010893024,0.00082485535,0.00092026807,0.00060059904,0.00057686295,0.0016368594,0.0015482538,0.001487836,0.0016235366],"category_scores_gemma":[0.0036050484,0.00059387484,0.0009847878,0.0006800616,0.00080293324,0.0016317493,0.0015868903,0.0013793055,0.00028267314],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009919397,0.000010920602,0.00022461849,0.000017399872,0.000007702032,0.00003493253,0.000034755183,0.9739836,0.0006210527,0.022362195,0.000203445,0.0024894471],"study_design_scores_gemma":[0.0000020359992,0.0000018981798,0.00001592549,0.0000016689062,9.531985e-7,0.0000034626642,0.0000030125002,0.99657345,0.00006094853,0.0031253865,0.0002096418,0.000001666279],"about_ca_topic_score_codex":0.015394976,"about_ca_topic_score_gemma":0.0070700664,"teacher_disagreement_score":0.015394976,"about_ca_system_score_codex":0.0014304768,"about_ca_system_score_gemma":0.0022995588,"threshold_uncertainty_score":0.03061074},"labels":[],"label_agreement":null},{"id":"W2163165845","doi":"10.1109/tse.2009.30","title":"Engineering of Framework-Specific Modeling Languages","year":2009,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Software engineering; Model-driven architecture; Domain-specific language; Reverse engineering; Domain (mathematical analysis); Modeling language; Programming language; Software development; Software","score_opus":0.024973816079868247,"score_gpt":0.26647242826469514,"score_spread":0.24149861218482688,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2163165845","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059769596,0.0001444442,0.98746383,0.00043174505,0.00005239096,0.00028181347,0.0001165231,0.0020516282,0.0034805376],"genre_scores_gemma":[0.028026251,0.00022145505,0.96883494,0.00009439021,0.000015475047,0.00028816215,0.00034974611,0.00065793423,0.0015117442],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9887494,0.0036954577,0.0011247005,0.00093715393,0.0046885773,0.0008048085],"domain_scores_gemma":[0.984445,0.0042068157,0.0012489414,0.005402988,0.004165584,0.00053077424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017439429,0.0013141355,0.00069093873,0.0026756946,0.0014519513,0.0048927017,0.0040556514,0.0017246742,0.0017291518],"category_scores_gemma":[0.027904933,0.0017619568,0.0031315102,0.0013518423,0.0023892329,0.005718647,0.005673054,0.0047546313,0.0007608156],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009181747,0.00025013738,0.0033707411,0.0006466237,0.00016286391,0.0007649044,0.004009055,0.07016752,0.01576685,0.7372467,0.007877974,0.15964492],"study_design_scores_gemma":[0.00011211535,0.00017098043,0.00095731183,0.0007720379,0.00021478091,0.0010738252,0.0011587285,0.39841312,0.04272996,0.23258252,0.32159507,0.00021962092],"about_ca_topic_score_codex":0.0060491846,"about_ca_topic_score_gemma":0.00894354,"teacher_disagreement_score":0.017439429,"about_ca_system_score_codex":0.0031105708,"about_ca_system_score_gemma":0.00834925,"threshold_uncertainty_score":0.092229605},"labels":[],"label_agreement":null},{"id":"W2167003754","doi":"10.1109/32.910857","title":"Design of multi-invariant data structures for robust shared accesses in multiprocessor systems","year":2001,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Distributed systems and fault tolerance","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Invariant (physics); Data structure; Multiprocessing; Fault tolerance; Algorithm; Distributed computing; Parallel computing; Theoretical computer science; Programming language; Mathematics","score_opus":0.07999694562695671,"score_gpt":0.27546634039445494,"score_spread":0.19546939476749825,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167003754","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028159706,0.00020897253,0.96961963,0.00010070723,0.000027707565,0.00013851379,0.000050309143,0.0011735641,0.000520829],"genre_scores_gemma":[0.3360291,0.00025458404,0.66128904,0.00014757046,0.00004550476,0.0005149794,0.00025142374,0.00034480353,0.0011230157],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9978828,0.0004734141,0.00038206336,0.0002586144,0.00079137745,0.00021170087],"domain_scores_gemma":[0.99540895,0.0013425095,0.00086614466,0.0012613784,0.0009872942,0.00013370771],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023269288,0.0004814156,0.0006083771,0.00093542185,0.0006983329,0.001426057,0.0021317645,0.000780124,0.00069009216],"category_scores_gemma":[0.004887075,0.0005692692,0.0006673943,0.00090624404,0.0014603151,0.0021756846,0.0013952883,0.0012236638,0.000318461],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006383609,0.00035521312,0.0052999835,0.00097349554,0.0001666809,0.0005349428,0.0011433415,0.26929238,0.15566058,0.26559526,0.005541082,0.29479873],"study_design_scores_gemma":[0.00018239638,0.00055072736,0.0005492224,0.000077789875,0.00011452794,0.00025906798,0.00011486911,0.75563407,0.16391143,0.057963148,0.020572204,0.00007059569],"about_ca_topic_score_codex":0.0007853579,"about_ca_topic_score_gemma":0.0010457804,"teacher_disagreement_score":0.0023269288,"about_ca_system_score_codex":0.00092607905,"about_ca_system_score_gemma":0.0014926055,"threshold_uncertainty_score":0.012306154},"labels":[],"label_agreement":null},{"id":"W2167110832","doi":"10.1109/32.852742","title":"Validating the ISO/IEC 15504 measure of software requirements analysis process capability","year":2000,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Software measurement; Process (computing); Software engineering; Process capability; Software Engineering Process Group; Software; Software quality; Software development process; Reliability engineering; Systems engineering; Software development; Work in process; Engineering; Operations management; Operating system","score_opus":0.021389347909023807,"score_gpt":0.26770558990715365,"score_spread":0.24631624199812985,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167110832","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85486466,0.0004451965,0.08211755,0.0008867293,0.00025065074,0.0030671635,0.001623273,0.00043246988,0.056312326],"genre_scores_gemma":[0.90256035,0.00025622032,0.088660434,0.00018701695,0.00004482012,0.0031198845,0.0030293209,0.0000962711,0.0020456812],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9370538,0.019631969,0.006947459,0.0019171658,0.033036616,0.0014130783],"domain_scores_gemma":[0.84297377,0.055727948,0.012448877,0.016962204,0.06970368,0.0021835994],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04564538,0.00063887046,0.00054193696,0.0055141277,0.0007761672,0.0019622853,0.0012708391,0.0010615506,0.001569815],"category_scores_gemma":[0.15471706,0.00032619966,0.0010813083,0.004732814,0.0013534513,0.0025824807,0.0020293728,0.0015712021,0.00089643605],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066820026,0.0037173019,0.48685753,0.00080676854,0.00035469103,0.00026560266,0.007209264,0.01611391,0.012891664,0.02777614,0.010627385,0.43271166],"study_design_scores_gemma":[0.00029613715,0.005902258,0.860414,0.0009547352,0.00016763264,0.0004950065,0.0058594104,0.04264056,0.0197738,0.009221892,0.054063704,0.00021091473],"about_ca_topic_score_codex":0.0037871231,"about_ca_topic_score_gemma":0.0036151765,"teacher_disagreement_score":0.04564538,"about_ca_system_score_codex":0.0022320184,"about_ca_system_score_gemma":0.005589456,"threshold_uncertainty_score":0.24139869},"labels":[],"label_agreement":null},{"id":"W2169782617","doi":"10.1109/tse.2002.1158285","title":"An operational process for goal-driven definition of measures","year":2002,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":148,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Empirical research; Set (abstract data type); Process (computing); Verifiable secret sharing; Software engineering; Management science; Programming language","score_opus":0.03966221107474621,"score_gpt":0.26917200462530516,"score_spread":0.22950979355055895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169782617","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00072710303,0.00017949686,0.99262685,0.0016336176,0.000100166726,0.0004068279,0.000112856906,0.00016638769,0.004046602],"genre_scores_gemma":[0.026703719,0.00021540969,0.9690554,0.0006048627,0.000101100966,0.0022152371,0.00030198204,0.00012063651,0.0006816884],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.96719915,0.018368095,0.003739479,0.0035811393,0.006305089,0.0008069801],"domain_scores_gemma":[0.9493905,0.025320232,0.0032466175,0.009038229,0.011549419,0.001455112],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.051249042,0.0026093507,0.0022272915,0.007595867,0.0032181372,0.009940111,0.006977259,0.004856401,0.005317696],"category_scores_gemma":[0.07780371,0.0014430253,0.0033917848,0.0062034084,0.01527843,0.01661984,0.00835169,0.010999686,0.002701184],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008925061,0.000033803568,0.000112770365,0.00010471282,0.000014464522,0.000056614743,0.0010421758,0.0010929777,0.00033072018,0.986704,0.00085105846,0.009647859],"study_design_scores_gemma":[0.000027672078,0.00007427421,0.00019903842,0.00025554287,0.000020697205,0.00012401825,0.00067452993,0.011137103,0.0008528931,0.9493262,0.037257023,0.00005110199],"about_ca_topic_score_codex":0.0024456216,"about_ca_topic_score_gemma":0.0014111458,"teacher_disagreement_score":0.051249042,"about_ca_system_score_codex":0.0041987393,"about_ca_system_score_gemma":0.008133901,"threshold_uncertainty_score":0.27103406},"labels":[],"label_agreement":null},{"id":"W2247133547","doi":"10.1109/tse.2015.2449319","title":"A Tool-Supported Methodology for Validation and Refinement of Early-Stage Domain Models","year":2015,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Domain (mathematical analysis); Unified Modeling Language; Class diagram; Software engineering; Domain model; Subject-matter expert; Set (abstract data type); Process (computing); Model-driven architecture; Domain engineering; Modeling language; Domain analysis; Use Case Diagram; Metamodeling; Requirements engineering; Artificial intelligence; Domain knowledge; Programming language; Software development; Software; Expert system","score_opus":0.08578734900011618,"score_gpt":0.2873067031678123,"score_spread":0.20151935416769612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2247133547","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029666855,0.000019647116,0.9939376,0.000058614743,0.000008951599,0.0003008449,0.00007175009,0.0023554,0.00028031695],"genre_scores_gemma":[0.024210772,0.000046081448,0.9740078,0.000039520528,0.000005593188,0.0005979678,0.000383049,0.00030527348,0.00040398643],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9834145,0.0075678453,0.0018398503,0.0018051818,0.0049220235,0.00045070986],"domain_scores_gemma":[0.9583037,0.019546408,0.0035628001,0.011741526,0.0064048017,0.0004407866],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020412963,0.0017342394,0.00092832505,0.0033097097,0.0009884137,0.0030536999,0.00331777,0.0015475301,0.0018968893],"category_scores_gemma":[0.06394171,0.001421181,0.0024069394,0.0015137604,0.0016091242,0.0036143113,0.0037185592,0.0032147786,0.0011435547],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031854174,0.0014317037,0.010702746,0.0019640462,0.00042111237,0.0009813705,0.008364727,0.07510893,0.1063266,0.082899064,0.0079597365,0.70352143],"study_design_scores_gemma":[0.00021988862,0.00093901675,0.004175024,0.0010512533,0.0002875672,0.0015303633,0.0014204641,0.7032342,0.13649178,0.05852066,0.091756545,0.0003731421],"about_ca_topic_score_codex":0.0017026748,"about_ca_topic_score_gemma":0.0023680665,"teacher_disagreement_score":0.020412963,"about_ca_system_score_codex":0.0013709246,"about_ca_system_score_gemma":0.004661141,"threshold_uncertainty_score":0.107955396},"labels":[],"label_agreement":null},{"id":"W2316930373","doi":"10.1109/tse.2016.2550458","title":"Developer Micro Interaction Metrics for Software Defect Prediction","year":2016,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Ministry of Education, Science and Technology; Neurosciences Research Foundation","keywords":"Computer science; Eclipse; Leverage (statistics); Software quality assurance; Software quality; Software bug; Software; Software metric; Software engineering; Source code; Software development; Task (project management); Plug-in; Machine learning; Operating system; Systems engineering","score_opus":0.0218172345002419,"score_gpt":0.25171140673187403,"score_spread":0.22989417223163214,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2316930373","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54180336,0.0028341431,0.42687723,0.00059440476,0.00013850188,0.0004286973,0.005089963,0.014388456,0.007845254],"genre_scores_gemma":[0.8668675,0.00027654937,0.12833352,0.0000695132,0.000044118453,0.00031360096,0.0028233684,0.0003589007,0.00091294496],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9939766,0.0015015847,0.00041319602,0.0006917517,0.003219349,0.00019746176],"domain_scores_gemma":[0.9585689,0.022108618,0.00883072,0.0038312296,0.0055983565,0.0010622416],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030977554,0.0015352081,0.00079647294,0.005848203,0.00035168862,0.0010307211,0.00087886484,0.0006483331,0.0013661777],"category_scores_gemma":[0.033821605,0.00037735724,0.0004354369,0.004054153,0.00027506636,0.0017621457,0.0010930753,0.0009973885,0.0005979152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048339934,0.0006256983,0.3305445,0.00051244907,0.00030385703,0.00017689088,0.0006222825,0.064542376,0.018905045,0.0034090856,0.010734591,0.56913984],"study_design_scores_gemma":[0.000057597153,0.0011131468,0.16426767,0.0001404713,0.00014771232,0.00035427234,0.00021815689,0.79847324,0.01909723,0.007268052,0.008717351,0.00014508466],"about_ca_topic_score_codex":0.002936218,"about_ca_topic_score_gemma":0.005377657,"teacher_disagreement_score":0.005848203,"about_ca_system_score_codex":0.00068939803,"about_ca_system_score_gemma":0.0008483418,"threshold_uncertainty_score":0.016382694},"labels":[],"label_agreement":null},{"id":"W2338921896","doi":"10.1109/tse.2016.2550441","title":"Test Case Prioritization Using Lexicographical Ordering","year":2016,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; University of Waterloo","keywords":"Lexicographical order; Computer science; Prioritization; Test case; Heuristic; Fault detection and isolation; Data mining; Fault (geology); Greedy algorithm; Reliability engineering; Algorithm; Machine learning; Artificial intelligence; Mathematics","score_opus":0.019273459654922546,"score_gpt":0.24197271955170754,"score_spread":0.222699259896785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2338921896","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03968177,0.00063365226,0.9466524,0.00041362012,0.00010983557,0.0011021936,0.00029924503,0.0023322226,0.008774976],"genre_scores_gemma":[0.21681851,0.00039904486,0.77866215,0.00024304162,0.00006595863,0.00040788465,0.0008610552,0.00035872156,0.002183746],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99458766,0.0017499481,0.0005448318,0.0005474344,0.0021919822,0.0003780957],"domain_scores_gemma":[0.98828596,0.006690451,0.00093275233,0.0013816061,0.0023577341,0.000351423],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021124894,0.0015072806,0.001344836,0.0060272943,0.001084929,0.0023874443,0.0016243877,0.00073615945,0.004009048],"category_scores_gemma":[0.01679645,0.00068780367,0.00094337517,0.0039748023,0.001042037,0.0015448299,0.0015018708,0.0014847069,0.0011699977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075218035,0.00069289905,0.005765572,0.0009322642,0.00016207338,0.00095831946,0.0008836312,0.105094776,0.050551347,0.038825557,0.008201285,0.78718007],"study_design_scores_gemma":[0.0004079413,0.0010146026,0.0036790725,0.00033087173,0.00025985035,0.0018881626,0.0007610202,0.80090374,0.06982267,0.09334195,0.027410148,0.0001800305],"about_ca_topic_score_codex":0.003542776,"about_ca_topic_score_gemma":0.006765392,"teacher_disagreement_score":0.0060272943,"about_ca_system_score_codex":0.0012816943,"about_ca_system_score_gemma":0.0037669935,"threshold_uncertainty_score":0.0134115815},"labels":[],"label_agreement":null},{"id":"W2339904014","doi":"10.1109/tse.2015.2487958","title":"Black-Box String Test Case Generation through a Multi-Objective Optimization","year":2015,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":53,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; String (physics); Random testing; Algorithm; Test case; Test (biology); Test functions for optimization; String searching algorithm; Optimization problem; Mathematics; Machine learning; Programming language; Data structure","score_opus":0.04811254978416045,"score_gpt":0.26542024025866834,"score_spread":0.21730769047450788,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2339904014","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046334747,0.0001853685,0.9501192,0.000117563046,0.000019938436,0.00026004566,0.00005651441,0.00078749115,0.002119124],"genre_scores_gemma":[0.43529698,0.00016028312,0.561384,0.00013520717,0.000015104285,0.0006177624,0.00021373376,0.00017864938,0.0019981724],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986695,0.0006117953,0.000066347675,0.00017954613,0.00037033335,0.000102472695],"domain_scores_gemma":[0.9963469,0.0026474055,0.00032877165,0.00016577494,0.00043211566,0.00007907952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019337004,0.0013275254,0.0009293341,0.001520473,0.00031509876,0.00068988174,0.001062233,0.001127921,0.002288584],"category_scores_gemma":[0.005252071,0.00049130537,0.000719195,0.0008271777,0.0006053451,0.00073930697,0.00097051874,0.00077963644,0.00034518115],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009628807,0.0001413607,0.0010317926,0.00012124499,0.00004255556,0.00019786842,0.00006351987,0.8965602,0.007137132,0.003969428,0.0006556612,0.08998288],"study_design_scores_gemma":[0.0000121624425,0.00007057523,0.00013110533,0.000008671258,0.000008645652,0.000029844785,0.0000074507084,0.99668866,0.0018181104,0.00095943746,0.0002611716,0.0000041338226],"about_ca_topic_score_codex":0.0012467526,"about_ca_topic_score_gemma":0.0010274397,"teacher_disagreement_score":0.002288584,"about_ca_system_score_codex":0.0005991488,"about_ca_system_score_gemma":0.0008723102,"threshold_uncertainty_score":0.010226488},"labels":[],"label_agreement":null},{"id":"W2472584751","doi":"10.1109/tse.2016.2586066","title":"A Study of Causes and Consequences of Client-Side JavaScript Bugs","year":2016,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Intel Corporation","keywords":"Unobtrusive JavaScript; JavaScript; Computer science; Rich Internet application; Document Object Model; Web application; Programmer; World Wide Web; Client-side; Ajax; Programming language; Web page","score_opus":0.024017693639851897,"score_gpt":0.2549549340538039,"score_spread":0.23093724041395203,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2472584751","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9948184,0.0007191637,0.0025437356,0.00024047833,0.000018342153,0.000119740995,0.00023879086,0.0002157596,0.0010855797],"genre_scores_gemma":[0.99671066,0.0003347187,0.0020822003,0.00007110564,0.000017470165,0.00005138793,0.00034122722,0.000046845456,0.00034443405],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9901273,0.00210985,0.0010746176,0.001285898,0.00475317,0.0006492412],"domain_scores_gemma":[0.7997758,0.120363146,0.049148068,0.00515491,0.023321213,0.0022369702],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046360707,0.0007427322,0.000613566,0.0077504274,0.0010631179,0.0011773563,0.0010202689,0.0010723185,0.0011777787],"category_scores_gemma":[0.0711334,0.00062970276,0.00080470985,0.0043458925,0.0011537491,0.002022413,0.0011139181,0.0010103182,0.0002711611],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023428966,0.00042760622,0.9391607,0.0004607798,0.000107321895,0.0025899021,0.0036962444,0.0014042608,0.003003351,0.00046808913,0.0007878635,0.047659647],"study_design_scores_gemma":[0.00002830602,0.00050414307,0.9792667,0.00018834067,0.00019822763,0.0036130643,0.0037411586,0.0073125735,0.0033617888,0.00054046384,0.0011988371,0.000046508434],"about_ca_topic_score_codex":0.0056458293,"about_ca_topic_score_gemma":0.0061083594,"teacher_disagreement_score":0.0077504274,"about_ca_system_score_codex":0.0012126681,"about_ca_system_score_gemma":0.0021552069,"threshold_uncertainty_score":0.024518132},"labels":[],"label_agreement":null},{"id":"W2474835145","doi":"10.1109/tse.2016.2584050","title":"An Empirical Comparison of Model Validation Techniques for Defect Prediction Models","year":2016,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":566,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Japan Society for the Promotion of Science; Japan Society for the Promotion of Science London; Compute Canada","keywords":"Computer science; Variance (accounting); Context (archaeology); Cross-validation; Model validation; Sample (material); Data mining; Predictive modelling; Software bug; Software; Machine learning","score_opus":0.054222375516789025,"score_gpt":0.33031650944663127,"score_spread":0.2760941339298422,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2474835145","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6999274,0.011148597,0.2750353,0.0014983312,0.00035505716,0.0005150843,0.0033843287,0.002021749,0.0061141704],"genre_scores_gemma":[0.9277461,0.001358534,0.06456732,0.00020037802,0.000074612486,0.0003637347,0.004694594,0.00037236148,0.000622304],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.953718,0.030661775,0.002978476,0.0045603006,0.007287162,0.0007942178],"domain_scores_gemma":[0.51112753,0.4312624,0.012200539,0.025861066,0.018464979,0.001083364],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.088046685,0.0020123732,0.0012534371,0.0054560574,0.001089895,0.0024070835,0.0021225763,0.0025900737,0.001157252],"category_scores_gemma":[0.26811898,0.00067469577,0.0030256722,0.004156694,0.0016456712,0.004996179,0.0021404293,0.003260061,0.0005979206],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022089372,0.000835041,0.26697707,0.002063969,0.0044846353,0.0002960939,0.0015419234,0.45862967,0.0025318018,0.015495986,0.011001714,0.2339331],"study_design_scores_gemma":[0.00020106128,0.0017242713,0.083267085,0.00085884816,0.0006721755,0.0006702605,0.00064923894,0.88682425,0.0038246422,0.015473312,0.0056385295,0.00019630369],"about_ca_topic_score_codex":0.004775952,"about_ca_topic_score_gemma":0.0060357214,"teacher_disagreement_score":0.91195333,"about_ca_system_score_codex":0.002420923,"about_ca_system_score_gemma":0.0019278629,"threshold_uncertainty_score":0.46564096},"labels":[],"label_agreement":null},{"id":"W2517486838","doi":"10.11368/tse.19.1","title":"Self－rewetting溶液を用いた自励振動型ヒートパイプの性能向上に関する研究(ブタノールとペンタノールの場合)","year":2011,"lang":"ja","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Thermodynamic Systems and Engines","field":"Engineering","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Programming language; Software engineering","score_opus":0.012023827778058813,"score_gpt":0.19080057488631014,"score_spread":0.1787767471082513,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2517486838","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7010796,0.0028308267,0.10151526,0.001041405,0.001357264,0.00019821349,0.00041268475,0.0019350789,0.1896296],"genre_scores_gemma":[0.8787457,0.00072062726,0.014359019,0.0003678814,0.00012531548,0.00007423209,0.00021475327,0.00045395893,0.104938634],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996309,0.000016702223,0.000011300288,0.00010854987,0.00016647497,0.000066000706],"domain_scores_gemma":[0.9996481,0.000050752824,0.000033928438,0.00012643315,0.00011237253,0.00002843372],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00023968298,0.00032363823,0.00032441792,0.0004235703,0.00092615496,0.0011638297,0.00050271227,0.00049124117,0.008747149],"category_scores_gemma":[0.0004559224,0.00038628394,0.0003921773,0.0002211457,0.0010427153,0.001918138,0.0008743773,0.0011241308,0.0025538346],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008074115,0.00017828538,0.0015074619,0.00024500664,0.000050419374,0.00056808145,0.0015343732,0.0019202228,0.782945,0.1392281,0.0065721804,0.06517019],"study_design_scores_gemma":[0.000014997687,0.00012012524,0.0031519611,0.00002477577,0.00003757961,0.00045444438,0.00043949854,0.010309249,0.8967858,0.020519642,0.068077035,0.000064824206],"about_ca_topic_score_codex":0.00065348006,"about_ca_topic_score_gemma":0.00087943353,"teacher_disagreement_score":0.008747149,"about_ca_system_score_codex":0.0003532919,"about_ca_system_score_gemma":0.00033915852,"threshold_uncertainty_score":0.029262185},"labels":[],"label_agreement":null},{"id":"W2530596726","doi":"10.1109/tse.2017.2671865","title":"Measuring the Impact of Code Dependencies on Software Architecture Recovery Techniques","year":2017,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":86,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Ontario Ministry of Research, Innovation and Science; National Science Foundation","keywords":"Computer science; Granularity; Architecture; Software; Software architecture; Software architecture description; Implementation; Reference architecture; Graph; Code (set theory); Distributed computing; Software engineering; Theoretical computer science; Programming language","score_opus":0.027883247811874713,"score_gpt":0.27180466622482546,"score_spread":0.24392141841295076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2530596726","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.885362,0.0014784344,0.10309223,0.00025207715,0.0000731609,0.00020415464,0.00062723685,0.0070555443,0.0018551663],"genre_scores_gemma":[0.8865195,0.00035654297,0.10997873,0.00006687454,0.000017731873,0.00012504263,0.0016071786,0.00062694174,0.0007014962],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9856695,0.0032600237,0.0012556417,0.0018969048,0.0071187983,0.00079923857],"domain_scores_gemma":[0.87981623,0.08321273,0.01118884,0.01572919,0.009383349,0.0006696154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005601871,0.001180549,0.00055388117,0.0028291296,0.0006103634,0.00076680304,0.0015153425,0.0009914893,0.0008156625],"category_scores_gemma":[0.090685554,0.00066809624,0.00080895994,0.00185802,0.0009452192,0.0028245621,0.0016106962,0.0018245269,0.00040646477],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013156764,0.0008209156,0.07999481,0.0012988121,0.00047861898,0.00042214216,0.0009820497,0.2815885,0.071679555,0.0018571776,0.0037370168,0.55582476],"study_design_scores_gemma":[0.00009122013,0.0020655296,0.08365214,0.00015554506,0.00041594292,0.00087531283,0.0005038338,0.749321,0.1544315,0.003006797,0.0053198743,0.00016139243],"about_ca_topic_score_codex":0.0029986552,"about_ca_topic_score_gemma":0.0041711554,"teacher_disagreement_score":0.005601871,"about_ca_system_score_codex":0.00081028923,"about_ca_system_score_gemma":0.0013798815,"threshold_uncertainty_score":0.029625893},"labels":[],"label_agreement":null},{"id":"W2530824252","doi":"10.1109/tse.2016.2616306","title":"A Framework for Evaluating the Results of the SZZ Approach for Identifying Bug-Introducing Changes","year":2016,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":190,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University; McGill University","funders":"","keywords":"Implementation; Computer science; Task (project management); Software bug; Software implementation; Software; Software engineering; Programming language; Systems engineering","score_opus":0.07051700283867052,"score_gpt":0.3266203963305439,"score_spread":0.2561033934918734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2530824252","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18175887,0.0011584981,0.7830005,0.0021481216,0.00022308582,0.0031583745,0.0072995215,0.010340092,0.010912854],"genre_scores_gemma":[0.41789708,0.00022154306,0.5741702,0.00030606065,0.0000618192,0.002099721,0.0040103984,0.00041689601,0.00081624935],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.91324455,0.034208298,0.009851528,0.007949765,0.032706924,0.0020389883],"domain_scores_gemma":[0.7617588,0.13147064,0.038196288,0.027574716,0.039200723,0.0017988258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08436929,0.0039090193,0.002666479,0.03423489,0.0021979169,0.007367328,0.0047058347,0.0032048132,0.0030984152],"category_scores_gemma":[0.23684873,0.0012255388,0.0031709776,0.012294005,0.0047069,0.008084452,0.0072175567,0.003259937,0.0012838282],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002032704,0.001332743,0.22531013,0.00316227,0.001773727,0.00048452243,0.006001778,0.13347536,0.020257903,0.05824849,0.018853698,0.52906674],"study_design_scores_gemma":[0.0008204114,0.0047197533,0.12419944,0.0008431581,0.000875353,0.0008194005,0.0039190142,0.7352008,0.033453107,0.07482974,0.019349,0.00097085745],"about_ca_topic_score_codex":0.010707685,"about_ca_topic_score_gemma":0.009767716,"teacher_disagreement_score":0.08436929,"about_ca_system_score_codex":0.003966857,"about_ca_system_score_gemma":0.004552765,"threshold_uncertainty_score":0.44619274},"labels":[],"label_agreement":null},{"id":"W2579161546","doi":"10.1109/tse.2017.2654244","title":"Using Natural Language Processing to Automatically Detect Self-Admitted Technical Debt","year":2017,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":220,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Technical debt; Computer science; Debt; Quality (philosophy); Bad debt; Code (set theory); Finance; Software development; Business; Software; Programming language; Set (abstract data type)","score_opus":0.016474308440274747,"score_gpt":0.29079089578154826,"score_spread":0.2743165873412735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2579161546","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54269034,0.0011948379,0.3868244,0.0026460127,0.000270334,0.00097418454,0.016264506,0.04292936,0.006205924],"genre_scores_gemma":[0.51156276,0.00048842287,0.44539315,0.000763437,0.00014629724,0.0008269796,0.036552895,0.0009254484,0.0033406785],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954224,0.0012123146,0.0006210185,0.00096207304,0.0016246103,0.00015760571],"domain_scores_gemma":[0.95295626,0.028992673,0.008670548,0.0023157916,0.006552402,0.0005124075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033610878,0.0011235254,0.0005177744,0.0055928575,0.0006546698,0.001574574,0.0014790483,0.0013845349,0.0010028395],"category_scores_gemma":[0.0233499,0.0005565571,0.000801995,0.0028632693,0.00073794526,0.0033241936,0.001387197,0.0016021637,0.0010208106],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009374428,0.0010531974,0.11374132,0.004000246,0.00021022647,0.005476706,0.006697209,0.014578611,0.16033259,0.0069390493,0.05151874,0.6345147],"study_design_scores_gemma":[0.00025091754,0.00061347574,0.08801452,0.00054865493,0.00023020108,0.0036057723,0.0041336655,0.73614323,0.07241608,0.03161733,0.06215944,0.00026674054],"about_ca_topic_score_codex":0.005790713,"about_ca_topic_score_gemma":0.00781743,"teacher_disagreement_score":0.005790713,"about_ca_system_score_codex":0.0011331933,"about_ca_system_score_gemma":0.0023017337,"threshold_uncertainty_score":0.017775357},"labels":[],"label_agreement":null},{"id":"W2754109047","doi":"10.1109/tse.2017.2750682","title":"Expanding Queries for Code Search Using Semantically Related API Class-names","year":2017,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Identifier; Programming language; Class (philosophy); Information retrieval; Natural language; Java; Code (set theory); Natural language user interface; World Wide Web; Natural language processing; Artificial intelligence","score_opus":0.039927966108913644,"score_gpt":0.30923921623353773,"score_spread":0.2693112501246241,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2754109047","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29748088,0.0054456163,0.5843075,0.0038294205,0.0003425226,0.0028516848,0.018412808,0.07399307,0.013336559],"genre_scores_gemma":[0.2913313,0.0012682987,0.6688223,0.001077013,0.00016160488,0.0007812011,0.030532772,0.0020394365,0.003986091],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99453545,0.0014162061,0.00064384757,0.0012735454,0.0017821575,0.00034871956],"domain_scores_gemma":[0.98820484,0.0074614594,0.00082477124,0.0011521028,0.0019869842,0.00036983305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026271895,0.0025123823,0.002023084,0.009228573,0.0013007214,0.0022660648,0.001906298,0.0022502698,0.0037364725],"category_scores_gemma":[0.018958285,0.0008667045,0.0019961589,0.0049021733,0.0011723455,0.006116499,0.003577225,0.0018436672,0.00258046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015095903,0.0014732654,0.037043616,0.0045444565,0.00035185475,0.0033135298,0.008927234,0.018697374,0.11824355,0.013744006,0.06959958,0.7225519],"study_design_scores_gemma":[0.00047765987,0.0011398977,0.030684577,0.0007037308,0.00064114784,0.0064420407,0.00820321,0.6208719,0.07731723,0.03078446,0.22229438,0.00043974057],"about_ca_topic_score_codex":0.011053401,"about_ca_topic_score_gemma":0.01491992,"teacher_disagreement_score":0.011053401,"about_ca_system_score_codex":0.001478537,"about_ca_system_score_gemma":0.0029249627,"threshold_uncertainty_score":0.02197814},"labels":[],"label_agreement":null},{"id":"W2757354086","doi":"10.1109/tse.2017.2756043","title":"A Study of Social Interactions in Open Source Component Use","year":2017,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Component (thermodynamics); Computer science; Open source; Java; Component-based software engineering; World Wide Web; Data science; Software; Software engineering; Software development; Programming language","score_opus":0.05336126301455479,"score_gpt":0.3085463286686244,"score_spread":0.2551850656540696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2757354086","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98888344,0.00021499235,0.0044498025,0.00045588927,0.000010713623,0.000066060165,0.00006231463,0.00001650328,0.0058403015],"genre_scores_gemma":[0.997943,0.00008597843,0.0012917506,0.000050300838,0.000015010132,0.000060384955,0.000058097346,0.000010287448,0.000485182],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.99133086,0.006191461,0.00026539862,0.000727631,0.0010200981,0.00046449687],"domain_scores_gemma":[0.9243405,0.053449497,0.012987045,0.00242696,0.003379578,0.0034165208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049584536,0.0003447232,0.0003797035,0.0028857286,0.003568445,0.002295165,0.00073609845,0.0010057234,0.0021131933],"category_scores_gemma":[0.03868699,0.00036920686,0.00034130985,0.002643157,0.002727826,0.0042817607,0.0034077454,0.0012509718,0.00030374824],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034524745,0.0007153734,0.67222327,0.00026995916,0.00018679953,0.0013992247,0.22722413,0.0015954598,0.004754935,0.017332759,0.0016887051,0.072264165],"study_design_scores_gemma":[0.00005054962,0.00042987632,0.8206862,0.00012874402,0.00009940687,0.0009917786,0.1275447,0.017605567,0.0011294559,0.0142373005,0.016984593,0.00011186681],"about_ca_topic_score_codex":0.0073265815,"about_ca_topic_score_gemma":0.008213068,"teacher_disagreement_score":0.0073265815,"about_ca_system_score_codex":0.0020499376,"about_ca_system_score_gemma":0.0013755072,"threshold_uncertainty_score":0.026223123},"labels":[],"label_agreement":null},{"id":"W2757935660","doi":"10.1109/tse.2017.2755005","title":"Revisiting the Performance Evaluation of Automated Approaches for the Retrieval of Duplicate Issue Reports","year":2017,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Information retrieval; Eclipse; Categorical variable; Software; Notation; Data mining; Machine learning; Programming language","score_opus":0.059809278371222435,"score_gpt":0.29953603310299953,"score_spread":0.2397267547317771,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2757935660","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76075083,0.039152455,0.11251883,0.0024396558,0.0022191436,0.0020370795,0.009971064,0.054041635,0.016869359],"genre_scores_gemma":[0.76335955,0.0033396718,0.19954818,0.0007670198,0.00057055627,0.0005544062,0.026106698,0.0014180174,0.004335886],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9527919,0.015054962,0.005800851,0.006950428,0.01757103,0.0018307389],"domain_scores_gemma":[0.8903194,0.060655896,0.006545838,0.018575722,0.02167656,0.002226623],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025950145,0.002935118,0.002799891,0.009422928,0.0017801761,0.0045992853,0.0051743933,0.0027004364,0.0019955833],"category_scores_gemma":[0.10055035,0.00075861765,0.0016422325,0.0066332724,0.0018039845,0.0073112487,0.002460938,0.002390605,0.0025070307],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031868487,0.00274815,0.03501901,0.00544623,0.0016704559,0.0005419428,0.0014648016,0.053426187,0.025611794,0.002721983,0.074530184,0.7936324],"study_design_scores_gemma":[0.0011415489,0.0056285677,0.076579966,0.0008575051,0.0012162229,0.0021394491,0.0025308405,0.7455607,0.07486661,0.0043726373,0.084446736,0.0006591755],"about_ca_topic_score_codex":0.033279806,"about_ca_topic_score_gemma":0.025601402,"teacher_disagreement_score":0.033279806,"about_ca_system_score_codex":0.0032798108,"about_ca_system_score_gemma":0.0051368275,"threshold_uncertainty_score":0.13723916},"labels":[],"label_agreement":null},{"id":"W2760127551","doi":"10.1109/tse.2017.2757480","title":"On the Use of Hidden Markov Model to Predict the Time to Fix Bugs","year":2017,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Hidden Markov model; Software bug; Software regression; Context (archaeology); Software; Markov model; Predictive modelling; Data mining; Software engineering; Software development; Markov chain; Data science; Software quality; Machine learning; Artificial intelligence; Programming language","score_opus":0.03385569789076816,"score_gpt":0.2492526278261282,"score_spread":0.21539692993536003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2760127551","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2555553,0.0025396084,0.7322895,0.0016186275,0.00022875937,0.00015906396,0.0011158654,0.0038874794,0.0026058317],"genre_scores_gemma":[0.87041736,0.0013454114,0.1232456,0.00030569948,0.00017899515,0.00010753046,0.0019209925,0.00012192202,0.0023565711],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989114,0.0004644015,0.00008749228,0.0002675962,0.00017059923,0.00009842088],"domain_scores_gemma":[0.9855887,0.012562917,0.0005554451,0.00036521786,0.0007701967,0.00015745785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004174621,0.0013218555,0.0010472619,0.0033396138,0.00071847875,0.0011353791,0.0011604715,0.0016228536,0.0010961357],"category_scores_gemma":[0.0112669235,0.0006017321,0.0015745126,0.0018900811,0.00032058515,0.00205049,0.00057972333,0.0017098605,0.0009012617],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077580113,0.0009304576,0.067306064,0.00020963767,0.0005695799,0.00034329316,0.0004082154,0.54484814,0.0035102847,0.003678328,0.0049602953,0.37245995],"study_design_scores_gemma":[0.0000073361816,0.000043019052,0.002042619,0.000015189691,0.000027561431,0.000036673355,0.000019075702,0.9962619,0.0003662425,0.0010048418,0.0001620954,0.000013475017],"about_ca_topic_score_codex":0.041752994,"about_ca_topic_score_gemma":0.03735791,"teacher_disagreement_score":0.041752994,"about_ca_system_score_codex":0.00083078415,"about_ca_system_score_gemma":0.0014708184,"threshold_uncertainty_score":0.08301991},"labels":[],"label_agreement":null},{"id":"W2793823312","doi":"10.1109/tse.2018.2810895","title":"Asymmetric Release Planning: Compromising Satisfaction against Dissatisfaction","year":2018,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Heuristics; Computer science; Completeness (order theory); Categorical variable; Feature (linguistics); Customer satisfaction; Stakeholder; Product (mathematics); Quality (philosophy); Process management; Operations research; Machine learning; Marketing; Mathematics; Business","score_opus":0.016881603055118155,"score_gpt":0.25352717358529264,"score_spread":0.2366455705301745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2793823312","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4701376,0.00045423172,0.51680887,0.0006541451,0.00005351672,0.0005035442,0.00014199324,0.00039744037,0.010848678],"genre_scores_gemma":[0.9376325,0.00010397645,0.060543846,0.00011567143,0.000015102846,0.00018261091,0.00008420687,0.00007409159,0.0012479852],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99654514,0.0016840448,0.0001869577,0.00044018336,0.0007093134,0.00043432257],"domain_scores_gemma":[0.99201196,0.004812964,0.0012794337,0.00066329003,0.0007453354,0.0004869228],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005231511,0.0014199506,0.00093373895,0.0008162891,0.00066683954,0.0018353226,0.0010642679,0.0009435451,0.0022163044],"category_scores_gemma":[0.01377782,0.0005453543,0.00063349964,0.0007166,0.000764375,0.0016669395,0.00135021,0.0012171471,0.0002807988],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010845499,0.0010005797,0.0118804285,0.00060596823,0.0002554173,0.00052311487,0.0009006875,0.749195,0.0382964,0.015689218,0.0019161471,0.17865258],"study_design_scores_gemma":[0.00015040529,0.0015963159,0.0063989065,0.00008020865,0.00015083764,0.00026120807,0.000656709,0.9613096,0.010543365,0.016829316,0.0019530726,0.000070071],"about_ca_topic_score_codex":0.0012488514,"about_ca_topic_score_gemma":0.0013778225,"teacher_disagreement_score":0.005231511,"about_ca_system_score_codex":0.00095002644,"about_ca_system_score_gemma":0.0016779118,"threshold_uncertainty_score":0.027667224},"labels":[],"label_agreement":null},{"id":"W2799550433","doi":"10.1109/tse.2018.2829722","title":"Integrative Double Kaizen Loop (IDKL): Towards a Culture of Continuous Learning and Sustainable Improvements for Software Organizations","year":2018,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Kaizen; Computer science; Software deployment; Process management; Productivity; Process (computing); Lean manufacturing; Workforce; Knowledge management; Operations management; Software engineering; Engineering","score_opus":0.007300316526246086,"score_gpt":0.24891636095248693,"score_spread":0.24161604442624085,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2799550433","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5276839,0.0017005972,0.3447903,0.023210092,0.000388531,0.0007187438,0.00006088368,0.00153959,0.09990732],"genre_scores_gemma":[0.8718513,0.00058736553,0.117119014,0.003546192,0.00004691972,0.0003487137,0.00004454738,0.00015654799,0.0062994845],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9870728,0.006754642,0.00045134366,0.0013344251,0.0035738596,0.000812936],"domain_scores_gemma":[0.9863599,0.0047894884,0.001990568,0.002643892,0.002047353,0.0021687956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0138137555,0.0005187332,0.00030353255,0.0013850532,0.0039350092,0.008536332,0.0023666192,0.0020474682,0.0012076981],"category_scores_gemma":[0.012716247,0.00048647658,0.00044480333,0.0011120405,0.012794403,0.010987752,0.011417585,0.0040496234,0.0004134304],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032874223,0.0013814119,0.024643365,0.0010902951,0.00010391309,0.0006800209,0.20472388,0.004952755,0.013803323,0.34062064,0.011039493,0.39663213],"study_design_scores_gemma":[0.00026014165,0.0023114486,0.023905393,0.002050126,0.00014730953,0.0030950664,0.13398944,0.027743721,0.015282901,0.3672662,0.42337283,0.00057537004],"about_ca_topic_score_codex":0.0015631806,"about_ca_topic_score_gemma":0.0025135763,"teacher_disagreement_score":0.0138137555,"about_ca_system_score_codex":0.005094929,"about_ca_system_score_gemma":0.010949632,"threshold_uncertainty_score":0.07305497},"labels":[],"label_agreement":null},{"id":"W2803395207","doi":"10.1109/tse.2018.2838131","title":"Use and Misuse of Continuous Integration Features: An Empirical Study of Projects That (Mis)Use Travis CI","year":2018,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Software deployment; Feature (linguistics); Software; Source code; Code (set theory); Software engineering; Programming language","score_opus":0.05555397349595297,"score_gpt":0.30736840377452834,"score_spread":0.25181443027857536,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2803395207","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9991161,0.0000612263,0.00041488896,0.000035690035,0.0000014112828,0.000014614204,0.00008840962,0.000016064345,0.0002515536],"genre_scores_gemma":[0.99802256,0.00009765325,0.0009862912,0.000027720795,0.00000468124,0.0000485651,0.0003577196,0.000025392124,0.0004295006],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9909128,0.0027804498,0.0012131592,0.0014588068,0.0030842384,0.0005504718],"domain_scores_gemma":[0.8561314,0.08041044,0.038211107,0.0073974603,0.014613304,0.0032363513],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0077846786,0.00035831126,0.0003362513,0.0031222622,0.00082943344,0.0013290169,0.00078760664,0.0007772349,0.0008021363],"category_scores_gemma":[0.056530483,0.0004231651,0.00026147984,0.0028844664,0.0013932197,0.0025252106,0.0017722,0.0010714469,0.0003842243],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015508357,0.00030703796,0.9522304,0.00014542365,0.000045104567,0.0005428843,0.01488865,0.00029961552,0.0019719317,0.0002145497,0.00048655967,0.028712794],"study_design_scores_gemma":[0.0000120884,0.00057964836,0.9783407,0.00007201128,0.000029245939,0.0011355433,0.01279694,0.002370702,0.0019921977,0.00020356308,0.0024314648,0.00003599395],"about_ca_topic_score_codex":0.002271158,"about_ca_topic_score_gemma":0.0033834856,"teacher_disagreement_score":0.99221534,"about_ca_system_score_codex":0.00064711424,"about_ca_system_score_gemma":0.0006972353,"threshold_uncertainty_score":0.041169822},"labels":[],"label_agreement":null},{"id":"W2803893519","doi":"10.1109/tse.2018.2836450","title":"Automatically Categorizing Software Technologies","year":2018,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Categorization; Software; Taxonomy (biology); Domain (mathematical analysis); Phrase; Software engineering; Programming language; Natural language processing; Artificial intelligence","score_opus":0.013858434838389265,"score_gpt":0.24211322902831106,"score_spread":0.22825479418992178,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2803893519","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2261531,0.0029250653,0.724333,0.0006666827,0.00038480788,0.001414602,0.014435169,0.013715809,0.015971793],"genre_scores_gemma":[0.27719933,0.0012755162,0.68350905,0.00019971713,0.00010827297,0.0014377659,0.030818602,0.0010296704,0.004422052],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9924078,0.0016468347,0.0011698277,0.0014026036,0.0029420375,0.00043100023],"domain_scores_gemma":[0.9766027,0.011469292,0.0024795195,0.002776133,0.006188128,0.00048426844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040957723,0.0013927474,0.001135503,0.03197317,0.0016499653,0.003374484,0.0017287715,0.0018544736,0.0026543115],"category_scores_gemma":[0.031396776,0.0005642247,0.0013240745,0.011563298,0.00077714375,0.008226857,0.0045831418,0.0012157346,0.0020838412],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030492365,0.0002298173,0.040111065,0.0037219222,0.00022142073,0.0011402427,0.007491906,0.004747843,0.050760537,0.035984628,0.021346934,0.8339387],"study_design_scores_gemma":[0.0001424075,0.00040912887,0.061425675,0.0026065477,0.000419847,0.0045920126,0.0146947075,0.32283607,0.10658189,0.13491079,0.3509709,0.0004100663],"about_ca_topic_score_codex":0.0031593374,"about_ca_topic_score_gemma":0.004679603,"teacher_disagreement_score":0.03197317,"about_ca_system_score_codex":0.0014894082,"about_ca_system_score_gemma":0.0030010296,"threshold_uncertainty_score":0.021660805},"labels":[],"label_agreement":null},{"id":"W2885308680","doi":"10.1109/tse.2018.2861735","title":"Leveraging Historical Associations between Requirements and Source Code to Identify Impacted Classes","year":2018,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Science Foundation","keywords":"Computer science; Set (abstract data type); Class (philosophy); Code smell; Intuition; Similarity (geometry); Semantic similarity; Source code; Data mining; Locality; Software; Information retrieval; Artificial intelligence; Software development; Software quality; Programming language","score_opus":0.05722962105556756,"score_gpt":0.32059560617191873,"score_spread":0.2633659851163512,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2885308680","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8881859,0.00078379764,0.100551955,0.00028190383,0.000049741142,0.00018836885,0.003072862,0.0029020698,0.0039834147],"genre_scores_gemma":[0.9374622,0.00024052252,0.055643473,0.000049383303,0.000048406535,0.00010284014,0.0054877084,0.00018450122,0.0007809915],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967895,0.00053772004,0.00034646937,0.0007093723,0.0014841422,0.00013273695],"domain_scores_gemma":[0.9666235,0.015117707,0.009384036,0.0025937757,0.0055270013,0.00075395685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022121156,0.00077404885,0.00052312156,0.008377003,0.00046762524,0.0010737951,0.0006206077,0.0006974104,0.0007181677],"category_scores_gemma":[0.026320703,0.0003050338,0.0007368044,0.0035869158,0.00038141332,0.0028402493,0.0009373887,0.0008100573,0.00055506954],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005011697,0.00044993425,0.5991451,0.0005187772,0.00032981558,0.00074597273,0.0011369151,0.034487575,0.019286083,0.0011120723,0.0035619927,0.3387246],"study_design_scores_gemma":[0.000031176416,0.00076773233,0.48953658,0.00010161412,0.00017297898,0.0010604905,0.0006884826,0.48340124,0.015244018,0.0027843437,0.0060953237,0.000115982126],"about_ca_topic_score_codex":0.0043524876,"about_ca_topic_score_gemma":0.010159703,"teacher_disagreement_score":0.008377003,"about_ca_system_score_codex":0.0006277714,"about_ca_system_score_gemma":0.00075364363,"threshold_uncertainty_score":0.011698961},"labels":[],"label_agreement":null},{"id":"W2904680236","doi":"10.1109/tse.2018.2884911","title":"A Study of Feature Scattering in the Linux Kernel","year":2018,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Vetenskapsrådet; Deutsche Forschungsgemeinschaft","keywords":"Code refactoring; Computer science; Feature (linguistics); Kernel (algebra); Code (set theory); Scattering; Software; Source code; Computer engineering; Operating system; Programming language; Optics; Physics; Set (abstract data type)","score_opus":0.033980790524803246,"score_gpt":0.284294568658101,"score_spread":0.25031377813329775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2904680236","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9976457,0.00014846031,0.0009878923,0.00021046605,0.0000026068046,0.000008989621,0.000011992021,0.0000107828255,0.000973148],"genre_scores_gemma":[0.998811,0.00011250739,0.0005993517,0.0000494658,0.0000042957518,0.000010528613,0.00003316087,0.000013501175,0.00036623763],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9957207,0.0018997715,0.00023786731,0.00045604212,0.001273544,0.00041213632],"domain_scores_gemma":[0.9445083,0.03356924,0.010166551,0.0022601536,0.0078111244,0.0016845998],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064337146,0.00023012045,0.00022801742,0.0020783423,0.0024028337,0.0019657193,0.0007602376,0.0006436016,0.00083725987],"category_scores_gemma":[0.034188353,0.00044249275,0.0002127881,0.0014909074,0.002553607,0.0047673155,0.0019841734,0.0014307097,0.00021173307],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000103549566,0.00033599514,0.55795485,0.0001230579,0.000032634598,0.0009091646,0.3742423,0.0006854288,0.0032057425,0.0030733421,0.0011684576,0.05816555],"study_design_scores_gemma":[0.00001510585,0.0008512899,0.6681326,0.00023524028,0.000045182933,0.0018567493,0.29396367,0.0075561176,0.0036682268,0.003466662,0.020082032,0.00012712087],"about_ca_topic_score_codex":0.0073612705,"about_ca_topic_score_gemma":0.007194142,"teacher_disagreement_score":0.0073612705,"about_ca_system_score_codex":0.002023007,"about_ca_system_score_gemma":0.0013810134,"threshold_uncertainty_score":0.034025133},"labels":[],"label_agreement":null},{"id":"W2909172538","doi":"10.1109/tse.2019.2891758","title":"The Impact of Correlated Metrics on the Interpretation of Defect Models","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Australian Research Council","keywords":"Consistency (knowledge bases); Ranking (information retrieval); Interpretation (philosophy); Computer science; Metric (unit); Data mining; Statistics; Machine learning; Artificial intelligence; Mathematics; Programming language","score_opus":0.014277181931630203,"score_gpt":0.2477259451862916,"score_spread":0.2334487632546614,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2909172538","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49013096,0.0030791068,0.49266303,0.0017900303,0.00030642765,0.00030613018,0.002049135,0.0029730906,0.0067021204],"genre_scores_gemma":[0.86897737,0.00038896248,0.12665683,0.00027228965,0.00008461954,0.00011606374,0.002318172,0.00077359116,0.00041193303],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.92465204,0.041829336,0.005449464,0.008062754,0.018974144,0.001032183],"domain_scores_gemma":[0.5389713,0.33771,0.037503302,0.057263017,0.027108101,0.0014443466],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.043280363,0.0020499502,0.001699579,0.006275942,0.0010460936,0.005392224,0.0018754132,0.0013643613,0.0010323805],"category_scores_gemma":[0.28535756,0.000909962,0.0022966275,0.0046757692,0.0026265462,0.0056612906,0.0032448731,0.0033409488,0.00047136802],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013962414,0.0004236394,0.34005594,0.0017917692,0.002403739,0.001022628,0.0032245712,0.23079541,0.015635068,0.026094012,0.007756543,0.36940044],"study_design_scores_gemma":[0.0002283128,0.0015514109,0.15296496,0.0009763682,0.0012094968,0.0018176361,0.0025238264,0.67530805,0.026802607,0.122178644,0.013958805,0.0004798193],"about_ca_topic_score_codex":0.0032307904,"about_ca_topic_score_gemma":0.004661975,"teacher_disagreement_score":0.95671964,"about_ca_system_score_codex":0.0022847326,"about_ca_system_score_gemma":0.0026562195,"threshold_uncertainty_score":0.2288912},"labels":[],"label_agreement":null},{"id":"W2909755202","doi":"10.1109/tse.2019.2893171","title":"Too Many User-Reviews! What Should App Developers Look at First?","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":86,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Polytechnique Montréal","keywords":"Key (lock); Computer science; Android (operating system); Mobile apps; World Wide Web; App store; Set (abstract data type); Android app; Star (game theory); Mobile device; Multimedia; Computer security","score_opus":0.023871746814439637,"score_gpt":0.2479435287887195,"score_spread":0.22407178197427988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2909755202","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9583822,0.0064423224,0.00727736,0.007325971,0.00049711886,0.00044776892,0.0032061024,0.0008745008,0.015546667],"genre_scores_gemma":[0.9819603,0.0014633794,0.0058118463,0.0021397753,0.00028310524,0.00020815413,0.0017178777,0.0001727244,0.006242947],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99455875,0.0019585474,0.0004797781,0.00065255136,0.002060124,0.0002902281],"domain_scores_gemma":[0.9491192,0.025369791,0.0082098525,0.0024736875,0.012648061,0.0021794455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003156778,0.0005860475,0.00068723987,0.0014337867,0.0010585921,0.0023247688,0.0004966543,0.001014656,0.0029581166],"category_scores_gemma":[0.04923952,0.00042334464,0.0005049134,0.0016493676,0.00046865857,0.0031814084,0.0007532365,0.0009158319,0.0028432868],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007060045,0.0004078387,0.6072073,0.0016782453,0.00033071468,0.0012732908,0.0168376,0.00045560522,0.008687019,0.0008590624,0.087299794,0.27425745],"study_design_scores_gemma":[0.000065510896,0.0007639156,0.8512921,0.0005927401,0.00026709528,0.0031110824,0.024289219,0.0051480993,0.004842173,0.0015103079,0.107933864,0.0001839892],"about_ca_topic_score_codex":0.0050200587,"about_ca_topic_score_gemma":0.011518411,"teacher_disagreement_score":0.0050200587,"about_ca_system_score_codex":0.0007288515,"about_ca_system_score_gemma":0.000665265,"threshold_uncertainty_score":0.016694844},"labels":[],"label_agreement":null},{"id":"W2914489208","doi":"10.1109/tse.2019.2897300","title":"Which Commits Can Be CI Skipped?","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Commit; Computer science; Process (computing); Java; Software engineering; Database; Programming language","score_opus":0.012768036516850528,"score_gpt":0.2325881247438001,"score_spread":0.2198200882269496,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914489208","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7270205,0.0029642438,0.23757413,0.0017067979,0.00072547205,0.00081303116,0.00404068,0.015859265,0.009295912],"genre_scores_gemma":[0.84151417,0.0009763623,0.14353952,0.00063253316,0.00014831728,0.00022850042,0.0071938736,0.0012145743,0.0045522805],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9893652,0.000993974,0.001331597,0.0027818398,0.0047090845,0.0008183306],"domain_scores_gemma":[0.93148464,0.028503474,0.013772416,0.009772255,0.014313019,0.002154173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007260547,0.0011997609,0.0009675661,0.003972981,0.0011746483,0.0026164425,0.0018220987,0.0011751266,0.0011160908],"category_scores_gemma":[0.059734862,0.0007372037,0.0011117825,0.0022193803,0.00077392685,0.0025487829,0.0015480474,0.0021779588,0.0009600546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00082930655,0.00034787395,0.47335035,0.0013355545,0.0003152367,0.0031534843,0.0024392908,0.019208593,0.036737606,0.0055555445,0.015528564,0.4411986],"study_design_scores_gemma":[0.00016065444,0.0014601502,0.30635396,0.0012813455,0.00089841284,0.0080000935,0.0055333306,0.46693006,0.10454127,0.027518233,0.07687652,0.0004459143],"about_ca_topic_score_codex":0.010920066,"about_ca_topic_score_gemma":0.01924554,"teacher_disagreement_score":0.010920066,"about_ca_system_score_codex":0.0009759408,"about_ca_system_score_gemma":0.0039979285,"threshold_uncertainty_score":0.03839791},"labels":[],"label_agreement":null},{"id":"W2940499664","doi":"10.1109/tse.2019.2912962","title":"The Mutation and Injection Framework: Evaluating Clone Detection Tools with Mutation Analysis","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"clone (Java method); Computer science; Java; Benchmark (surveying); Mutation; Precision and recall; Mutation testing; Programming language; Data mining; Artificial intelligence; Genetics; Biology; Gene","score_opus":0.014663928079709844,"score_gpt":0.26347687219838184,"score_spread":0.248812944118672,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2940499664","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5235961,0.0025166513,0.42416802,0.000444264,0.00020224809,0.0013419758,0.0019476702,0.041070763,0.004712358],"genre_scores_gemma":[0.5910744,0.0003215819,0.40239242,0.00022199548,0.00004515392,0.00052524154,0.0032743025,0.0011025344,0.0010424269],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98024565,0.0059648217,0.0017276466,0.0029023017,0.008373496,0.0007860214],"domain_scores_gemma":[0.96493495,0.017439043,0.005249152,0.0047995597,0.006593419,0.0009838006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013809586,0.0026109682,0.0013727602,0.010633336,0.0008772677,0.0024567728,0.0036039215,0.0029190304,0.0008057837],"category_scores_gemma":[0.048152328,0.00065107987,0.0018591292,0.003790109,0.0017145191,0.0039145933,0.0025326165,0.0015805727,0.00045868155],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017874695,0.0024105401,0.10266031,0.0023219415,0.001694006,0.0007079409,0.001539988,0.20177193,0.084197946,0.012250714,0.010927849,0.5777294],"study_design_scores_gemma":[0.0002169547,0.0028302567,0.025759598,0.00018491788,0.00028770103,0.0007466789,0.0003063138,0.88250834,0.07660377,0.0039696535,0.0063597662,0.00022611952],"about_ca_topic_score_codex":0.008690738,"about_ca_topic_score_gemma":0.0056877374,"teacher_disagreement_score":0.013809586,"about_ca_system_score_codex":0.00259093,"about_ca_system_score_gemma":0.0029939546,"threshold_uncertainty_score":0.073032975},"labels":[],"label_agreement":null},{"id":"W2946233956","doi":"10.1109/tse.2019.2918520","title":"Characterizing Crowds to Better Optimize Worker Recommendation in Crowdsourced Testing","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"National Key Research and Development Program of China; China Scholarship Council; National Natural Science Foundation of China","keywords":"Computer science; Crowds; Crowdsourcing; Task (project management); Context (archaeology); Software bug; Relevance (law); Test (biology); Machine learning; Software; Data science; Computer security; World Wide Web; Engineering","score_opus":0.01883341979320734,"score_gpt":0.24014374368048746,"score_spread":0.2213103238872801,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946233956","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16749662,0.0032160548,0.81617427,0.0013169921,0.00027263697,0.0008239317,0.00088982296,0.004227981,0.005581686],"genre_scores_gemma":[0.8155667,0.0006038613,0.17698951,0.0006349104,0.00017905585,0.0005656869,0.0013756427,0.00031602627,0.0037685204],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.9970777,0.0009854118,0.00015636731,0.0009774049,0.00054352987,0.00025950721],"domain_scores_gemma":[0.99048525,0.005975623,0.00079924625,0.0009795956,0.001122556,0.0006377077],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034388425,0.002437201,0.0026087153,0.0019063995,0.0011175921,0.0015434986,0.0032568104,0.0021177912,0.0021235915],"category_scores_gemma":[0.016308047,0.00091825705,0.0011010173,0.0016038334,0.0009977957,0.0022367279,0.0019158508,0.0012528821,0.0010752406],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012775373,0.0011055141,0.026799142,0.0011001528,0.00040802415,0.0005775124,0.0010129936,0.49781248,0.013002403,0.003870812,0.014634171,0.43839926],"study_design_scores_gemma":[0.00012101016,0.00024384398,0.0028614448,0.00005864278,0.000086518696,0.00010407621,0.0002612292,0.984938,0.0020773546,0.0061388835,0.0030654576,0.00004349176],"about_ca_topic_score_codex":0.016239008,"about_ca_topic_score_gemma":0.018285027,"teacher_disagreement_score":0.016239008,"about_ca_system_score_codex":0.0012552363,"about_ca_system_score_gemma":0.0022207724,"threshold_uncertainty_score":0.03228897},"labels":[],"label_agreement":null},{"id":"W2950127321","doi":"10.1109/tse.2019.2921343","title":"What Do Programmers Discuss About Blockchain? A Case Study on the Use of Balanced LDA and the Reference Architecture of a Domain to Capture Online Discussions About Blockchain Platforms Across Stack Exchange Communities","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Fundamental Research Funds for the Central Universities; National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Computer science; Popularity; Domain (mathematical analysis); Blockchain; Stack (abstract data type); Architecture; Data science; World Wide Web; Data mining; Computer security; Operating system","score_opus":0.025508109890508716,"score_gpt":0.259146886751046,"score_spread":0.2336387768605373,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950127321","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9678026,0.00041042396,0.015331476,0.0021937399,0.00003403136,0.00016797599,0.0007689438,0.0002540523,0.013036762],"genre_scores_gemma":[0.98026836,0.0002550036,0.01213443,0.00042797695,0.00004510461,0.00019639524,0.001174175,0.00015457856,0.005343963],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9953484,0.0033421454,0.00012469638,0.00044577962,0.00046325903,0.0002757091],"domain_scores_gemma":[0.9741796,0.019453363,0.0017351446,0.0013111399,0.002223301,0.0010974894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005565516,0.000354064,0.00027191988,0.0028775379,0.003950443,0.0027018662,0.0006993718,0.00124621,0.0021161183],"category_scores_gemma":[0.019911144,0.00023708533,0.00024360162,0.004838145,0.0015483769,0.006619885,0.0020678323,0.0014649553,0.00079458655],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012339503,0.0009935369,0.2643462,0.0010261328,0.00008693611,0.0056255898,0.4275269,0.0057509537,0.018630665,0.035790503,0.025484303,0.21350427],"study_design_scores_gemma":[0.0001818773,0.0005386688,0.1945128,0.000687884,0.00009912711,0.0026028184,0.4121505,0.09291046,0.014817676,0.03513832,0.2460932,0.00026666842],"about_ca_topic_score_codex":0.009818797,"about_ca_topic_score_gemma":0.020512192,"teacher_disagreement_score":0.009818797,"about_ca_system_score_codex":0.0020061373,"about_ca_system_score_gemma":0.001364434,"threshold_uncertainty_score":0.029433548},"labels":[],"label_agreement":null},{"id":"W2951710749","doi":"10.1109/tse.2019.2924006","title":"Locating Latent Design Information in Developer Discussions: A Study on Pull Requests","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Université de Montréal; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Maintainability; Classifier (UML); Documentation; Software engineering; Software design; Machine learning; Software; Robustness (evolution); Source lines of code; Artificial intelligence; Data mining; Software development; Programming language","score_opus":0.021608475185581517,"score_gpt":0.24979844873978285,"score_spread":0.22818997355420134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2951710749","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.994009,0.00018931473,0.003926924,0.00018283108,0.000010074053,0.000060034192,0.00023087517,0.000107106345,0.0012838924],"genre_scores_gemma":[0.99321884,0.0001466108,0.003761887,0.00013154709,0.000032260745,0.00010732825,0.0010035459,0.000099012985,0.001499072],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9900956,0.0059546134,0.0007297847,0.00095446245,0.0018936532,0.00037184317],"domain_scores_gemma":[0.7545282,0.20457844,0.017771086,0.0067908634,0.014059922,0.0022714643],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009141678,0.00045879395,0.00052441284,0.004007635,0.0016352152,0.0017757871,0.0008543476,0.0014571769,0.0013321654],"category_scores_gemma":[0.09755081,0.00036611754,0.00041472173,0.0025459128,0.0010535562,0.0040167435,0.0018201839,0.0016531975,0.00086731795],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014355687,0.0022477326,0.74849766,0.0008249818,0.000135155,0.0015044395,0.094988994,0.0012562734,0.016018867,0.001511595,0.004763383,0.12681545],"study_design_scores_gemma":[0.00013750218,0.0021606428,0.8552497,0.00036682285,0.00014957282,0.0027406942,0.057276353,0.04226146,0.012826707,0.003572237,0.023058092,0.0002002378],"about_ca_topic_score_codex":0.002327137,"about_ca_topic_score_gemma":0.0038675515,"teacher_disagreement_score":0.009141678,"about_ca_system_score_codex":0.0009655558,"about_ca_system_score_gemma":0.0005779243,"threshold_uncertainty_score":0.0483464},"labels":[],"label_agreement":null},{"id":"W2954796040","doi":"10.1109/tse.2019.2925345","title":"What's Wrong with My Benchmark Results? Studying Bad Practices in JMH Benchmarks","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Vetenskapsrådet","keywords":"Java; Computer science; Benchmark (surveying); Software engineering; Open source; Statement (logic); Best practice; Empirical research; Software; Programming language; Data science","score_opus":0.01793363133666667,"score_gpt":0.24993281964352593,"score_spread":0.23199918830685926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2954796040","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84823877,0.006395219,0.08478143,0.029764337,0.0017643078,0.0004449777,0.002255945,0.012397343,0.013957748],"genre_scores_gemma":[0.90653825,0.001374499,0.07916202,0.004560225,0.0003046464,0.00028524984,0.0017063181,0.003973196,0.002095669],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.94298553,0.016762555,0.0055161836,0.0064613787,0.025804428,0.002469903],"domain_scores_gemma":[0.72016096,0.13211912,0.038443673,0.04546117,0.059254583,0.0045604073],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.035781264,0.0014890275,0.001219866,0.0054103937,0.0020380071,0.005827317,0.0029314377,0.0019458486,0.0010280415],"category_scores_gemma":[0.25754267,0.0011182447,0.0010691839,0.006552599,0.0036007501,0.007134066,0.0025579275,0.003448797,0.0008943212],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079016475,0.00091602997,0.49943078,0.0017064677,0.0009916914,0.0012264898,0.012757837,0.015293237,0.022995869,0.009340058,0.052005358,0.38254604],"study_design_scores_gemma":[0.000367405,0.0031146395,0.5499659,0.004591341,0.0016915618,0.004978426,0.021066418,0.105749615,0.13830219,0.050796162,0.118205875,0.0011704547],"about_ca_topic_score_codex":0.006064502,"about_ca_topic_score_gemma":0.008025615,"teacher_disagreement_score":0.96421874,"about_ca_system_score_codex":0.0028411963,"about_ca_system_score_gemma":0.0021502164,"threshold_uncertainty_score":0.1892317},"labels":[],"label_agreement":null},{"id":"W2958385005","doi":"10.1109/tse.2019.2927908","title":"Methodological Principles for Reproducible Performance Evaluation in Cloud Computing","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Cloud Computing and Resource Management","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Electronic Components and Systems for European Leadership; Deutsche Forschungsgemeinschaft; Stiftelsen för Strategisk Forskning; Knowledge Foundation; Oracle; National Science Foundation","keywords":"Cloud computing; Computer science; Data science; Benchmark (surveying); Set (abstract data type); Systematic review; Open research; Domain (mathematical analysis); Field (mathematics); World Wide Web; MEDLINE","score_opus":0.08977160147076024,"score_gpt":0.29942080426299666,"score_spread":0.20964920279223642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2958385005","genre_codex":"methods","genre_gemma":"methods","domain_codex":"methods","domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004031004,0.007672615,0.9380499,0.011559708,0.0023057642,0.026839217,0.0010106127,0.00050123443,0.0080299],"genre_scores_gemma":[0.06762575,0.0028773947,0.83186257,0.005546292,0.0011843196,0.08903112,0.0004950855,0.00033639686,0.0010409502],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.19265483,0.66571116,0.06781175,0.01844428,0.05321197,0.0021659262],"domain_scores_gemma":[0.13349943,0.56931543,0.06215487,0.15536171,0.077638686,0.0020298993],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7030904,0.0040853876,0.0051087076,0.011276845,0.0061878827,0.016277652,0.009943871,0.009710663,0.0042752987],"category_scores_gemma":[0.8078597,0.0029048352,0.0077949692,0.011384421,0.02152878,0.013402607,0.01290246,0.012447186,0.0029427581],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023394416,0.0012479251,0.016300168,0.04670658,0.0051184767,0.00056551164,0.021131014,0.011575374,0.005476195,0.5790975,0.019414676,0.2910271],"study_design_scores_gemma":[0.003133638,0.0039273603,0.012538623,0.04735649,0.0034806582,0.00060729834,0.007540326,0.015753986,0.017263386,0.72365713,0.16393363,0.00080744363],"about_ca_topic_score_codex":0.0030747047,"about_ca_topic_score_gemma":0.0023726693,"teacher_disagreement_score":0.29690957,"about_ca_system_score_codex":0.008415102,"about_ca_system_score_gemma":0.033609662,"threshold_uncertainty_score":0.3661424},"labels":[],"label_agreement":null},{"id":"W2964175311","doi":"10.1109/tse.2018.2868349","title":"Debugging Static Analysis","year":2018,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Universität Paderborn; École Polytechnique Fédérale de Lausanne; Heinz Nixdorf Stiftung; Deutsche Forschungsgemeinschaft","keywords":"Static analysis; Debugging; Computer science; Debugger; Static program analysis; Programming language; Program analysis; Algorithmic program debugging; Software bug; Source code; Data-flow analysis; Call graph; Code (set theory); Tracing; Taint checking; Software engineering; Data flow diagram; Database; Software; Software development","score_opus":0.013345475541958063,"score_gpt":0.25049080371884513,"score_spread":0.23714532817688708,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964175311","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07586462,0.0021104568,0.8193345,0.002171436,0.00072713714,0.0005529908,0.0032068528,0.061371952,0.03466006],"genre_scores_gemma":[0.5319442,0.0017231868,0.4349337,0.0010973373,0.00026332805,0.00042130981,0.005686294,0.009968057,0.013962522],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9883937,0.0042959102,0.00093524257,0.0019887574,0.0037346806,0.00065156695],"domain_scores_gemma":[0.9344354,0.031865686,0.00394537,0.01234233,0.01662031,0.0007909361],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012373361,0.002106892,0.0010445262,0.005723947,0.0015190635,0.0030419189,0.0018104868,0.0012164471,0.011575365],"category_scores_gemma":[0.06857033,0.0010182904,0.0008696718,0.0029602393,0.0009999969,0.0057580895,0.0027174861,0.0017754643,0.0059984955],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077093655,0.00029055093,0.03547023,0.0012904486,0.00015620772,0.00072383846,0.008347537,0.008312818,0.02270024,0.028518658,0.065960094,0.82745844],"study_design_scores_gemma":[0.00021848634,0.0007992882,0.027574575,0.0023363638,0.0005216845,0.0033632726,0.0061707483,0.1314228,0.0955759,0.07363362,0.6579229,0.00046036957],"about_ca_topic_score_codex":0.0027682881,"about_ca_topic_score_gemma":0.0030214149,"teacher_disagreement_score":0.012373361,"about_ca_system_score_codex":0.0010700546,"about_ca_system_score_gemma":0.003261547,"threshold_uncertainty_score":0.06543738},"labels":[],"label_agreement":null},{"id":"W2972246936","doi":"10.1109/tse.2019.2938525","title":"Enabling Good Work Habits in Software Developers through Reflective Goal-Setting","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Personal Information Management and User Behavior","field":"Decision Sciences","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Work (physics); Productivity; Reflection (computer programming); Goal setting; Knowledge management; Set (abstract data type); Software; Domain (mathematical analysis); Psychology; Engineering","score_opus":0.07449546207864395,"score_gpt":0.34146741393617197,"score_spread":0.26697195185752803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2972246936","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91249585,0.00015861297,0.078830585,0.0007490064,0.000026154808,0.00060418725,0.000030883184,0.00077754166,0.006327313],"genre_scores_gemma":[0.91541535,0.00012661809,0.082502715,0.00017410137,0.000010027082,0.00048128868,0.000063826,0.000058226742,0.0011677278],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9919565,0.005941221,0.0003195052,0.00053851446,0.00076879427,0.00047549352],"domain_scores_gemma":[0.97079235,0.0164114,0.0032433718,0.0046241726,0.0027808037,0.0021479372],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009830686,0.00061428745,0.0003192313,0.0012373462,0.00097567856,0.0022318955,0.0010060597,0.0008351901,0.00089410564],"category_scores_gemma":[0.035478074,0.0005641045,0.00054095074,0.00044291542,0.0010388409,0.0016829579,0.0027908005,0.0012008774,0.0003692491],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040214934,0.01080554,0.19955476,0.00086889847,0.00016917635,0.0004096186,0.06761533,0.004907044,0.025419226,0.004116105,0.004241686,0.6814905],"study_design_scores_gemma":[0.0011697236,0.01051124,0.6334978,0.001979687,0.0006687098,0.0019219896,0.07832678,0.07065827,0.07572246,0.04947824,0.07527631,0.0007887333],"about_ca_topic_score_codex":0.0010346107,"about_ca_topic_score_gemma":0.0026285595,"teacher_disagreement_score":0.009830686,"about_ca_system_score_codex":0.00077109516,"about_ca_system_score_gemma":0.0024452205,"threshold_uncertainty_score":0.05199027},"labels":[],"label_agreement":null},{"id":"W2973035781","doi":"10.1109/tse.2019.2948910","title":"CrySL: An Extensible Approach to Validating the Correct Usage of Cryptographic APIs","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Heinz Nixdorf Stiftung","keywords":"Computer science; Cryptography; Java; Android (operating system); Programming language; Algorithm; Operating system","score_opus":0.011485362405329838,"score_gpt":0.22432560316352004,"score_spread":0.2128402407581902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2973035781","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00701,0.00007452381,0.85223556,0.00037783885,0.00018653076,0.0013789068,0.0018107877,0.133322,0.0036038782],"genre_scores_gemma":[0.08874052,0.00035261124,0.8618627,0.0010983393,0.00012631463,0.0016173769,0.008511724,0.02925565,0.008434723],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.96341294,0.00990898,0.009001705,0.0038505779,0.01217076,0.0016550914],"domain_scores_gemma":[0.9016521,0.034326922,0.009582249,0.035339233,0.017468598,0.0016308896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04197494,0.003852508,0.0013934041,0.0059030047,0.0020136484,0.007625946,0.00800713,0.00415478,0.012848999],"category_scores_gemma":[0.083945945,0.0054793293,0.004593388,0.0015835288,0.0048937257,0.01620341,0.011356145,0.0081822695,0.0072538033],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026186006,0.0021400114,0.04686948,0.004733159,0.00085921265,0.0043939217,0.009973049,0.07285585,0.09737901,0.23437026,0.109471776,0.41433573],"study_design_scores_gemma":[0.00046322425,0.0009986287,0.0048011583,0.0017819415,0.00037775832,0.0023222587,0.0015723802,0.38800982,0.20854975,0.05972672,0.33039668,0.0009997173],"about_ca_topic_score_codex":0.011154454,"about_ca_topic_score_gemma":0.009703412,"teacher_disagreement_score":0.04197494,"about_ca_system_score_codex":0.0030385284,"about_ca_system_score_gemma":0.008323564,"threshold_uncertainty_score":0.22198737},"labels":[],"label_agreement":null},{"id":"W2973296283","doi":"10.1109/tse.2019.2941880","title":"Studying the Impact of Noises in Build Breakage Data","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Replicate; Breakage; Server; Troubleshooting; Timeout; Data mining; Data science; World Wide Web; Statistics; Operating system","score_opus":0.028169490543929115,"score_gpt":0.2874072299796261,"score_spread":0.25923773943569695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2973296283","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92641056,0.0022963672,0.046055168,0.0014548409,0.0002719241,0.0002468513,0.019779481,0.0014928672,0.0019919882],"genre_scores_gemma":[0.9183797,0.0005034966,0.027362704,0.0004347799,0.00015422616,0.00021745767,0.05196543,0.00026876468,0.0007133158],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97417724,0.00901694,0.0027713676,0.0068251453,0.0060557583,0.0011535463],"domain_scores_gemma":[0.78962535,0.16123354,0.017718391,0.021536443,0.00810322,0.0017830126],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02222134,0.0017119925,0.0011884323,0.005724118,0.0011744557,0.0030294706,0.0022314962,0.0025414096,0.00076428434],"category_scores_gemma":[0.120219305,0.0010696872,0.0022155938,0.0072757746,0.0022832735,0.0048682513,0.002670206,0.003545919,0.0006528511],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063003576,0.00049468956,0.8586049,0.00065742765,0.00071562914,0.0007393794,0.0024418714,0.08708007,0.0023728972,0.0027629368,0.00686986,0.036630183],"study_design_scores_gemma":[0.00011995005,0.00050338905,0.60936385,0.00047540216,0.0005280126,0.001469314,0.0034593863,0.33902267,0.007897339,0.010545905,0.02637993,0.00023473905],"about_ca_topic_score_codex":0.018862357,"about_ca_topic_score_gemma":0.020744963,"teacher_disagreement_score":0.02222134,"about_ca_system_score_codex":0.001913406,"about_ca_system_score_gemma":0.0018311989,"threshold_uncertainty_score":0.11751908},"labels":[],"label_agreement":null},{"id":"W2975871742","doi":"10.1109/tse.2019.2942301","title":"Smart Contract Development: Challenges and Opportunities","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":706,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsoft (Canada)","funders":"National Key Research and Development Program of China; Nanjing University; National Natural Science Foundation of China","keywords":"Computer science; Software engineering; Engineering management; Computer security; Data science; Engineering","score_opus":0.02204994130856623,"score_gpt":0.20349607014277102,"score_spread":0.1814461288342048,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2975871742","genre_codex":"commentary","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3693969,0.027097281,0.0671446,0.44652727,0.0011882928,0.00037088778,0.00024747627,0.00036171082,0.087665595],"genre_scores_gemma":[0.927071,0.016793747,0.03694543,0.009340804,0.0005449488,0.00037064907,0.00027148082,0.00017113674,0.008490859],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9612517,0.023548767,0.0017789677,0.0022884968,0.007927803,0.0032042435],"domain_scores_gemma":[0.8889309,0.0805716,0.007070479,0.0034977898,0.010973083,0.008956217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.029797642,0.00047259667,0.00045859368,0.0017567547,0.006974207,0.011396775,0.002860813,0.0056933067,0.003684186],"category_scores_gemma":[0.055367734,0.00075456133,0.0005808386,0.0028272725,0.009169787,0.02179913,0.007713047,0.006003411,0.0009809256],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013857505,0.00057507097,0.04425592,0.002211343,0.000038251565,0.006093742,0.17250487,0.0033395183,0.002457255,0.34385192,0.048330422,0.37620306],"study_design_scores_gemma":[0.00003383427,0.00018181304,0.014244301,0.002801368,0.000022795395,0.0058376,0.4023587,0.01221249,0.0017254615,0.17276056,0.38763073,0.00019027523],"about_ca_topic_score_codex":0.0039022958,"about_ca_topic_score_gemma":0.005650427,"teacher_disagreement_score":0.029797642,"about_ca_system_score_codex":0.0048341122,"about_ca_system_score_gemma":0.014706432,"threshold_uncertainty_score":0.15758687},"labels":[],"label_agreement":null},{"id":"W2989443621","doi":"10.1109/tse.2019.2952130","title":"An Empirical Study of Dependency Downgrades in the npm Ecosystem","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Downgrade; Dependency (UML); Software versioning; Reuse; Software; Software engineering; Computer security; Operating system","score_opus":0.01786606815314012,"score_gpt":0.28024781358299794,"score_spread":0.2623817454298578,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2989443621","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9949597,0.00023213061,0.0012926115,0.0004376889,0.000010617639,0.00005878188,0.00026416717,0.00005192393,0.002692236],"genre_scores_gemma":[0.99719954,0.00017606006,0.0014416801,0.00009295469,0.000014269019,0.000048794573,0.0005006304,0.00003750519,0.0004885683],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9886565,0.004677258,0.0013342127,0.0014916646,0.003129233,0.00071114185],"domain_scores_gemma":[0.6695696,0.19338502,0.08991594,0.01883767,0.022716327,0.0055754716],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.015534879,0.0003573726,0.0003171775,0.0029444417,0.0014929866,0.0029682384,0.0014983389,0.001106839,0.0029059625],"category_scores_gemma":[0.15066732,0.00058486365,0.0004217536,0.0033948438,0.0021856835,0.007066883,0.0022975479,0.0026265027,0.00056177744],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014225743,0.00038891967,0.9388191,0.0002576807,0.000059445094,0.00069094996,0.028902097,0.0004645716,0.0006988836,0.0017988522,0.0017718843,0.026005333],"study_design_scores_gemma":[0.000018982944,0.00023400123,0.9516799,0.00022972815,0.000052055922,0.0007951373,0.030175764,0.0042097755,0.0007516078,0.0016141052,0.01018018,0.000058847065],"about_ca_topic_score_codex":0.006871186,"about_ca_topic_score_gemma":0.0066953213,"teacher_disagreement_score":0.9844651,"about_ca_system_score_codex":0.0019994036,"about_ca_system_score_gemma":0.0014356675,"threshold_uncertainty_score":0.082157254},"labels":[],"label_agreement":null},{"id":"W2995892549","doi":"10.1109/tse.2019.2960357","title":"Effects of Personality Traits on Pull Request Acceptance","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Personality Traits and Psychology","field":"Psychology","cited_by":48,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Big Five personality traits; Personality; Human–computer interaction; Software engineering; Psychology; Social psychology","score_opus":0.012923066489828869,"score_gpt":0.26799891989481184,"score_spread":0.25507585340498296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2995892549","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.997326,0.0000633459,0.0009146767,0.000116905394,0.0000110499805,0.00002303651,0.00009917886,0.000040670755,0.0014050801],"genre_scores_gemma":[0.99880517,0.000024807203,0.00038333557,0.000022126334,0.000008316388,0.000017500433,0.00009104046,0.000013595416,0.0006341545],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9955682,0.0019062214,0.00039720273,0.00040943024,0.0013317047,0.00038722897],"domain_scores_gemma":[0.91572064,0.04658848,0.02108012,0.0046137976,0.0067268214,0.0052701957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049657426,0.00044066846,0.00025295248,0.00091289944,0.0004911588,0.0016555829,0.00028236202,0.00040433055,0.0027806591],"category_scores_gemma":[0.043998316,0.00022346708,0.0005512483,0.00065077806,0.00043005703,0.0009072829,0.0009828181,0.00088438805,0.00078491913],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014661047,0.00013073337,0.98343116,0.000026574817,0.00005657174,0.0001121862,0.0012493798,0.000170533,0.00064977387,0.000047154135,0.00022111545,0.01375818],"study_design_scores_gemma":[0.0000035874339,0.0000923946,0.9977568,0.000007358381,0.000014383207,0.00007637795,0.0007437463,0.00073545193,0.00023724722,0.000055467674,0.00026778362,0.000009409915],"about_ca_topic_score_codex":0.0020753844,"about_ca_topic_score_gemma":0.0022708976,"teacher_disagreement_score":0.0049657426,"about_ca_system_score_codex":0.00040227213,"about_ca_system_score_gemma":0.00037209227,"threshold_uncertainty_score":0.026261628},"labels":[],"label_agreement":null},{"id":"W2999284478","doi":"10.1109/tse.2020.2966994","title":"<i>checsdm</i>: A Method for Ensuring Consistency in Heterogeneous Safety-Critical System Design","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Consistency (knowledge bases); Set (abstract data type); Engineering design process; Component (thermodynamics); Software engineering; Programming language; Artificial intelligence; Engineering","score_opus":0.028589741221183025,"score_gpt":0.25684404679384637,"score_spread":0.22825430557266335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2999284478","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00044350067,0.000027293127,0.9975114,0.00010476463,0.000024737836,0.0001978139,0.0000549147,0.0008981223,0.0007373854],"genre_scores_gemma":[0.010314692,0.000062669074,0.9873487,0.00013338888,0.000022367329,0.00047345427,0.00031429142,0.00042513284,0.0009053418],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9710179,0.012028962,0.0029815342,0.003330106,0.0097135445,0.00092805456],"domain_scores_gemma":[0.9636341,0.015098688,0.002091807,0.011715546,0.006810931,0.0006488993],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025255112,0.0020484175,0.0018957693,0.0041966583,0.0028064062,0.0060941153,0.0057538953,0.003306881,0.0054851035],"category_scores_gemma":[0.0617539,0.0024704118,0.0040547037,0.0033331618,0.0043237354,0.0060033114,0.00924519,0.005538142,0.002251604],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035080838,0.0003948635,0.0029851077,0.0015745197,0.00041795603,0.0009222064,0.0029689448,0.086264715,0.019412465,0.3865637,0.025358949,0.47278568],"study_design_scores_gemma":[0.00035253167,0.0003201363,0.0007620213,0.0008896272,0.00034140208,0.00096907624,0.00064607116,0.57368124,0.06316043,0.1968185,0.16181198,0.00024703256],"about_ca_topic_score_codex":0.004880357,"about_ca_topic_score_gemma":0.0059676473,"teacher_disagreement_score":0.025255112,"about_ca_system_score_codex":0.0032355231,"about_ca_system_score_gemma":0.010456473,"threshold_uncertainty_score":0.1335634},"labels":[],"label_agreement":null},{"id":"W3001461005","doi":"10.1109/tse.2020.2967380","title":"A Machine Learning Approach to Improve the Detection of CI Skip Commits","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Commit; Computer science; Machine learning; Facilitator; Decision tree; Artificial intelligence; Tree (set theory); Software; Popularity; Data mining; Software engineering; Database; Operating system","score_opus":0.01601098347251498,"score_gpt":0.21960679703534175,"score_spread":0.20359581356282677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3001461005","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43202573,0.0020081345,0.5425647,0.0017149051,0.00046226443,0.0006231099,0.003948612,0.011754049,0.00489842],"genre_scores_gemma":[0.8252323,0.00021736146,0.16818781,0.00024112668,0.00014276954,0.0002516361,0.0038253646,0.000074876225,0.001826753],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99659854,0.00053967495,0.00055422547,0.0010148507,0.00085871154,0.0004340363],"domain_scores_gemma":[0.98483014,0.0071334722,0.0023233213,0.00094643945,0.004294467,0.00047207097],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035416132,0.0014408735,0.0012936488,0.008532834,0.00080309424,0.0015024858,0.001996175,0.00153225,0.0010546292],"category_scores_gemma":[0.014653332,0.00030022053,0.0009999755,0.0041996865,0.0004294133,0.0015758182,0.0008356221,0.002014086,0.0011155652],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043368968,0.0016132161,0.18168782,0.00033279645,0.00029681987,0.0007884841,0.00031985287,0.062693655,0.010493075,0.0014095287,0.012703349,0.7272276],"study_design_scores_gemma":[0.00002001765,0.00019658078,0.018183578,0.000044694876,0.00006850734,0.000312341,0.00014153782,0.9705283,0.0068273204,0.0017111327,0.0019271838,0.000038689708],"about_ca_topic_score_codex":0.00884106,"about_ca_topic_score_gemma":0.008269885,"teacher_disagreement_score":0.00884106,"about_ca_system_score_codex":0.00094023946,"about_ca_system_score_gemma":0.0017709008,"threshold_uncertainty_score":0.018730104},"labels":[],"label_agreement":null},{"id":"W3002536493","doi":"10.1109/tse.2020.2968061","title":"<i>MoMIT</i>: Porting a JavaScript Interpreter on a Quarter Coin","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Parallel Computing and Optimization Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Porting; JavaScript; Interpreter; Programming language; Unobtrusive JavaScript; Quarter (Canadian coin); Operating system; Software engineering; World Wide Web; Rich Internet application; Software","score_opus":0.01290825836886121,"score_gpt":0.2126756645624824,"score_spread":0.1997674061936212,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3002536493","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015862275,0.00012585412,0.9517081,0.00025976362,0.00010978133,0.00016612823,0.00020382252,0.01728418,0.014280072],"genre_scores_gemma":[0.12931432,0.00016788757,0.85652894,0.00034823662,0.00003403068,0.00029693017,0.0006853279,0.004010913,0.0086133545],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999398,0.00012379073,0.00006334567,0.00010169131,0.00025070526,0.000062481275],"domain_scores_gemma":[0.9993243,0.00023251053,0.000097406,0.00012610204,0.00017634989,0.000043361553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008722396,0.0011937363,0.0004665367,0.00047938514,0.00043072377,0.0012388717,0.0014010412,0.00083769957,0.0062401704],"category_scores_gemma":[0.0032484112,0.00043930006,0.0007997524,0.0004521654,0.0005973191,0.0014765644,0.0008500394,0.0011582931,0.0020610539],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006498614,0.00040915795,0.0050584166,0.0009437482,0.00016447455,0.0005709696,0.00057314534,0.18198536,0.18510605,0.053813186,0.035495605,0.53523004],"study_design_scores_gemma":[0.0000958731,0.00017664378,0.0006323004,0.00008248785,0.00006392399,0.00022473537,0.00007779397,0.7919243,0.14397267,0.0055714604,0.057111926,0.00006595366],"about_ca_topic_score_codex":0.0018506513,"about_ca_topic_score_gemma":0.0022027118,"teacher_disagreement_score":0.0062401704,"about_ca_system_score_codex":0.00063391734,"about_ca_system_score_gemma":0.00086591707,"threshold_uncertainty_score":0.020875394},"labels":[],"label_agreement":null},{"id":"W3006039974","doi":"10.1109/tse.2020.2973997","title":"ConfigMiner: Identifying the Appropriate Configuration Options for Config-Related User Questions by Mining Online Forums","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Task (project management); Rank (graph theory); Software; State (computer science); World Wide Web; Information retrieval; Data science; Programming language; Engineering; Systems engineering","score_opus":0.02648402744795977,"score_gpt":0.26287446803427095,"score_spread":0.23639044058631117,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3006039974","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.50239027,0.004200014,0.29989877,0.001851212,0.0002906933,0.0017904658,0.03412071,0.14622231,0.009235489],"genre_scores_gemma":[0.7116198,0.0006108514,0.2442765,0.00072008977,0.00014212083,0.0011914824,0.034924537,0.0021073166,0.00440734],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99569076,0.0016662066,0.0003723279,0.001051867,0.0010210895,0.00019768858],"domain_scores_gemma":[0.9784386,0.016293138,0.0018568527,0.001661145,0.0012157677,0.0005345025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003935153,0.0027034509,0.0010993655,0.0047783325,0.00053533056,0.0015861146,0.0016987094,0.0017226755,0.0036160615],"category_scores_gemma":[0.021621117,0.0005563741,0.0009364822,0.0012418655,0.00046993926,0.004137794,0.0016362866,0.0011635771,0.003108296],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021979301,0.0018509811,0.122926906,0.0033517177,0.00032708078,0.002035319,0.0042471145,0.011152651,0.044014018,0.0026573292,0.08544515,0.71979374],"study_design_scores_gemma":[0.00044214845,0.001994992,0.10418809,0.00079532654,0.00032387523,0.004908059,0.0043585333,0.7026992,0.06505839,0.01712997,0.09764477,0.00045669934],"about_ca_topic_score_codex":0.0008316546,"about_ca_topic_score_gemma":0.0024849747,"teacher_disagreement_score":0.0047783325,"about_ca_system_score_codex":0.0005138492,"about_ca_system_score_gemma":0.00062569516,"threshold_uncertainty_score":0.020811379},"labels":[],"label_agreement":null},{"id":"W3014214759","doi":"10.1109/tse.2020.2983399","title":"Studying Ad Library Integration Strategies of Top Free-to-Download Apps","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Download; Computer science; World Wide Web; App store; Mobile apps; Revenue; Set (abstract data type)","score_opus":0.013909002787377889,"score_gpt":0.21969032033219593,"score_spread":0.20578131754481804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3014214759","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99350834,0.0010266622,0.0019096214,0.00007120041,0.000014755326,0.00008875861,0.0008888834,0.000329062,0.0021626179],"genre_scores_gemma":[0.9849407,0.00082446984,0.007832004,0.00006052622,0.000024985693,0.00008935254,0.0032847903,0.00016158869,0.0027815604],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99747914,0.00032662376,0.0003108621,0.00046633763,0.0011345885,0.00028239388],"domain_scores_gemma":[0.98365605,0.008706247,0.0031912953,0.0009975157,0.0030663821,0.00038256758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015876293,0.0008071701,0.00063291425,0.008586183,0.0007295228,0.0018254354,0.0006239613,0.00053667574,0.00076229277],"category_scores_gemma":[0.012773959,0.00043704288,0.0007540362,0.0040474413,0.0005067992,0.002473056,0.00087556784,0.00063315587,0.00076227717],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036066424,0.00043237547,0.80185735,0.0005641203,0.00023293629,0.0015341231,0.004346715,0.0019973638,0.009052375,0.00071140577,0.003657159,0.17525347],"study_design_scores_gemma":[0.000019768742,0.00025872086,0.9276349,0.00020101927,0.00035944846,0.0030070462,0.0035046332,0.04077015,0.011914139,0.0008141394,0.011407355,0.0001087201],"about_ca_topic_score_codex":0.009050507,"about_ca_topic_score_gemma":0.013406548,"teacher_disagreement_score":0.009050507,"about_ca_system_score_codex":0.00053146074,"about_ca_system_score_gemma":0.000721119,"threshold_uncertainty_score":0.017995656},"labels":[],"label_agreement":null},{"id":"W3017224552","doi":"10.1109/tse.2020.2984086","title":"Detecting Developers’ Task Switches and Types","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Personal Information Management and User Behavior","field":"Decision Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Task (project management); Software engineering; Programming language; Systems engineering","score_opus":0.14610293055338866,"score_gpt":0.3287958043560832,"score_spread":0.18269287380269453,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3017224552","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97247815,0.00026558078,0.021261696,0.00011905179,0.000056351506,0.000358178,0.0019378796,0.0011279129,0.0023951698],"genre_scores_gemma":[0.96435183,0.0001665189,0.029949052,0.00007012918,0.000026528043,0.00046104105,0.0030789494,0.0001322252,0.0017636678],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9900686,0.0018795715,0.001416724,0.0026904948,0.003408847,0.00053573964],"domain_scores_gemma":[0.9055823,0.04787508,0.021449663,0.0070830164,0.015331995,0.00267787],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056725163,0.00095269363,0.00063365,0.006633297,0.0005952839,0.0021918903,0.001374239,0.00093779207,0.00083300803],"category_scores_gemma":[0.055922598,0.0006220698,0.00048892247,0.0021973168,0.00047413015,0.0016524147,0.0015276347,0.0010492856,0.0007087276],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000529482,0.00029747043,0.855206,0.00038722326,0.00012102988,0.00045043151,0.007838023,0.001847968,0.012377226,0.00042019947,0.0037009902,0.11682395],"study_design_scores_gemma":[0.000059298614,0.00050149456,0.92469525,0.00017637691,0.00010416584,0.000708505,0.005541159,0.045883458,0.012685926,0.0013367094,0.008153163,0.00015452638],"about_ca_topic_score_codex":0.006414209,"about_ca_topic_score_gemma":0.011641994,"teacher_disagreement_score":0.006633297,"about_ca_system_score_codex":0.0010675123,"about_ca_system_score_gemma":0.0013018239,"threshold_uncertainty_score":0.029999495},"labels":[],"label_agreement":null},{"id":"W3025537827","doi":"10.1109/tse.2021.3087087","title":"Generating Unit Tests for Documentation","year":2021,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Documentation; Internal documentation; Unit testing; Computer science; Software documentation; Redundancy (engineering); Artifact (error); Source code; Software engineering; Software; Database; Programming language; Operating system; Software development; Artificial intelligence; Software development process; Software construction","score_opus":0.03303154572484931,"score_gpt":0.29791213454178017,"score_spread":0.2648805888169309,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3025537827","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.096995,0.00033441148,0.8557764,0.00028487298,0.00020577968,0.0007329653,0.001840688,0.03698016,0.0068497336],"genre_scores_gemma":[0.30035293,0.00019444784,0.6826017,0.0002303398,0.00007078141,0.0007251151,0.0056124646,0.006055742,0.004156501],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99159086,0.0028214757,0.000867408,0.0011295043,0.0032139353,0.00037680118],"domain_scores_gemma":[0.92666346,0.04100972,0.0051042875,0.0151686575,0.0112155685,0.00083830376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004975728,0.0014381963,0.0010553538,0.0042161867,0.00050228596,0.0020287924,0.0020131979,0.0014484953,0.0059088278],"category_scores_gemma":[0.05983199,0.00093481835,0.0012752995,0.002070534,0.00077377044,0.001818247,0.0020320544,0.0011553483,0.002950515],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007447159,0.00064666447,0.022070628,0.0012290626,0.00018790046,0.0017485793,0.0012860312,0.04636938,0.05480761,0.023118045,0.025785325,0.82200617],"study_design_scores_gemma":[0.00043511565,0.0010599777,0.007760834,0.0005326889,0.0002315466,0.0027963032,0.00041267605,0.54669374,0.3340314,0.03947902,0.06634304,0.00022368548],"about_ca_topic_score_codex":0.00092455465,"about_ca_topic_score_gemma":0.0010407611,"teacher_disagreement_score":0.0059088278,"about_ca_system_score_codex":0.0007804861,"about_ca_system_score_gemma":0.0017661856,"threshold_uncertainty_score":0.026314437},"labels":[],"label_agreement":null},{"id":"W3032170634","doi":"10.1109/tse.2020.2998503","title":"Automatic Generation of Acceptance Test Cases From Use Case Specifications: An NLP-Based Approach","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"European Research Council; European Commission","keywords":"Computer science; Executable; Test case; Software requirements specification; Acceptance testing; Software engineering; Test script; System under test; Conformance testing; Test Management Approach; Test (biology); Formal specification; Reliability engineering; Software; Software system; Programming language; Machine learning; Software construction","score_opus":0.11804880023702771,"score_gpt":0.2662858943771059,"score_spread":0.1482370941400782,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032170634","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018216228,0.00012319794,0.96683425,0.00018535838,0.000033443557,0.0007333611,0.00060079165,0.010876961,0.0023962602],"genre_scores_gemma":[0.15627919,0.00021447147,0.8348013,0.00019829486,0.000027954362,0.0011143199,0.003714979,0.0018287658,0.0018207087],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9922091,0.003222575,0.00055735477,0.0010467963,0.0026175063,0.00034670354],"domain_scores_gemma":[0.97451514,0.018245103,0.0019695573,0.002090059,0.0029402403,0.00023982154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040108664,0.002351195,0.0009876164,0.003688694,0.0007067601,0.0022708867,0.0023693317,0.00205882,0.004802352],"category_scores_gemma":[0.024978606,0.0013680565,0.002370307,0.0014788869,0.0013522003,0.0015896691,0.0020797905,0.0017525019,0.0021299284],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070131524,0.0012489951,0.009812379,0.0020404777,0.0003739326,0.005674814,0.0036217067,0.24068002,0.10256836,0.03426008,0.013172232,0.58584577],"study_design_scores_gemma":[0.00014874198,0.00021771003,0.0013108127,0.00020354285,0.00014051993,0.001003115,0.0003551022,0.91228515,0.056300826,0.012903058,0.015046919,0.00008457332],"about_ca_topic_score_codex":0.0037135738,"about_ca_topic_score_gemma":0.0038034425,"teacher_disagreement_score":0.004802352,"about_ca_system_score_codex":0.0011407291,"about_ca_system_score_gemma":0.0019599276,"threshold_uncertainty_score":0.021211743},"labels":[],"label_agreement":null},{"id":"W3037099619","doi":"10.1109/tse.2020.3004525","title":"Why Do Software Developers Use Static Analysis Tools? A User-Centered Study of Developer Needs and Motivations","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":76,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Heinz Nixdorf Stiftung; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Usability; Software engineering; Software development; Software; Static program analysis; Secure coding; World Wide Web; Human–computer interaction; Software security assurance; Computer security; Programming language; Information security","score_opus":0.04324661639113619,"score_gpt":0.24984638888479016,"score_spread":0.20659977249365397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037099619","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91442186,0.0020586904,0.039648477,0.018315265,0.00011475785,0.00015350689,0.000103665974,0.00055006205,0.024633806],"genre_scores_gemma":[0.9831381,0.00071159005,0.011073067,0.0019499101,0.00004112423,0.000118888696,0.00007261216,0.00020689382,0.0026878868],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.96248454,0.021245819,0.0017916844,0.0022950177,0.009986603,0.0021962563],"domain_scores_gemma":[0.7920291,0.14190274,0.015734402,0.0075311484,0.037223857,0.0055786474],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02803061,0.0008497949,0.00057800306,0.0052990187,0.003396101,0.006828937,0.0017762918,0.0035153849,0.0013875637],"category_scores_gemma":[0.12884767,0.0016153859,0.00047759473,0.002468004,0.004267436,0.011321725,0.0034341624,0.0023317325,0.0009285348],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026645794,0.00043001122,0.30056092,0.0005924513,0.00011180793,0.0017116584,0.4993075,0.0004421394,0.0086118905,0.02619916,0.00800587,0.15376006],"study_design_scores_gemma":[0.00014922905,0.0006244833,0.22668,0.002227695,0.00027631884,0.0059570377,0.5423213,0.014821085,0.010991545,0.05685009,0.13839242,0.00070875575],"about_ca_topic_score_codex":0.004187184,"about_ca_topic_score_gemma":0.0066067907,"teacher_disagreement_score":0.97196937,"about_ca_system_score_codex":0.0022186844,"about_ca_system_score_gemma":0.0037094322,"threshold_uncertainty_score":0.14824176},"labels":[],"label_agreement":null},{"id":"W3042598532","doi":"10.1109/tse.2020.3008850","title":"Execution of Partial State Machine Models","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Partial evaluation; Static analysis; Execution model; Unified Modeling Language; Programming language; Program transformation; Model transformation; Semantics (computer science); Software; Overhead (engineering); Source code; Software development; Transformation (genetics); Artificial intelligence","score_opus":0.019574979642059568,"score_gpt":0.20920851662593612,"score_spread":0.18963353698387655,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3042598532","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051020246,0.0002070066,0.9326855,0.00028002117,0.00005176088,0.00023813502,0.0006108742,0.0051064338,0.009799996],"genre_scores_gemma":[0.5956313,0.00048043896,0.39157993,0.00014495075,0.000034796212,0.00061906973,0.0025572975,0.0007202373,0.008231916],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99650687,0.0011379938,0.00024152691,0.0005064913,0.0012380767,0.00036908622],"domain_scores_gemma":[0.99316674,0.0036955501,0.00038259354,0.0017958635,0.00083243067,0.0001267159],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026784563,0.0009307603,0.000863025,0.0010461186,0.00078374904,0.002809345,0.0018207645,0.0011687635,0.00493109],"category_scores_gemma":[0.009594225,0.0009083123,0.0023101962,0.0007746573,0.001682679,0.0035206021,0.002491386,0.0014408345,0.0008374534],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004012574,0.0001825856,0.0058216113,0.0005523897,0.00018886861,0.0010794512,0.0013882533,0.61753285,0.014727758,0.2839275,0.0023953149,0.071802095],"study_design_scores_gemma":[0.0000389452,0.00014624668,0.00043557072,0.00007692743,0.00008136764,0.00013022007,0.000106223044,0.88779783,0.015406512,0.08520454,0.01053179,0.000043920343],"about_ca_topic_score_codex":0.005827381,"about_ca_topic_score_gemma":0.006970341,"teacher_disagreement_score":0.005827381,"about_ca_system_score_codex":0.0013160362,"about_ca_system_score_gemma":0.002894055,"threshold_uncertainty_score":0.016496122},"labels":[],"label_agreement":null},{"id":"W3088218866","doi":"10.1109/tse.2020.3025732","title":"Automated Generation of Consistent Graph Models With Multiplicity Reasoning","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Nemzeti Kutatási, Fejlesztési és Innovaciós Alap; Natural Sciences and Engineering Research Council of Canada; Nemzeti Kutatási Fejlesztési és Innovációs Hivatal; Innovációs és Technológiai Minisztérium; Emberi Eroforrások Minisztériuma","keywords":"Computer science; Solver; Theoretical computer science; Predicate abstraction; Graph; Abstraction; Programming language; Model checking","score_opus":0.030861502862934462,"score_gpt":0.2177657069829095,"score_spread":0.18690420411997505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3088218866","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016387306,0.000041026065,0.98082125,0.00011305924,0.000012604342,0.00008139158,0.00019190965,0.0012975904,0.0010538637],"genre_scores_gemma":[0.2258062,0.000118694414,0.7713407,0.000091516704,0.000009603036,0.00020294497,0.0009510929,0.00046303257,0.0010162314],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977952,0.00084570673,0.000120694756,0.0003373621,0.00078070513,0.00012017343],"domain_scores_gemma":[0.9949898,0.0031995012,0.00027874002,0.0010035535,0.000460262,0.00006816995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020732789,0.0009032535,0.0007143156,0.001151155,0.0005072677,0.0013477376,0.0017676078,0.0011042152,0.0030299718],"category_scores_gemma":[0.010409748,0.00071958645,0.0019902654,0.000921758,0.0013258032,0.0026789976,0.0024922867,0.0014990589,0.0004881602],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012570508,0.0001099462,0.0019388206,0.00031642482,0.00009353661,0.0006815997,0.0005079218,0.7135571,0.017100569,0.14478122,0.0026889131,0.11809824],"study_design_scores_gemma":[0.00003399743,0.00003607922,0.00011911814,0.000026967917,0.000030278368,0.00009580626,0.000076035205,0.91004854,0.009783727,0.076234855,0.0034987992,0.000015937816],"about_ca_topic_score_codex":0.0028316248,"about_ca_topic_score_gemma":0.0057335515,"teacher_disagreement_score":0.0030299718,"about_ca_system_score_codex":0.0010094532,"about_ca_system_score_gemma":0.002082944,"threshold_uncertainty_score":0.010964632},"labels":[],"label_agreement":null},{"id":"W3091299024","doi":"10.1109/tse.2020.3027255","title":"Comparing Block-Based Programming Models for Two-Armed Robots","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Teaching and Learning Programming","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Robot; Human–computer interaction; Task (project management); Block (permutation group theory); Robotics; Programming by demonstration; Artificial intelligence; Inductive programming; Programming paradigm; Software engineering; Programming language; Systems engineering; Engineering","score_opus":0.039325971948369134,"score_gpt":0.24808900354820956,"score_spread":0.20876303159984044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3091299024","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34068358,0.0004191373,0.646668,0.00040000046,0.000093626586,0.0004898707,0.0003427675,0.0012057377,0.009697355],"genre_scores_gemma":[0.7388131,0.0004466827,0.2543935,0.000119153505,0.000016948869,0.0008884364,0.00060235464,0.00040334492,0.004316444],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973266,0.0016075431,0.0001916861,0.00024240707,0.00047657848,0.0001551602],"domain_scores_gemma":[0.97644764,0.019818967,0.0009075683,0.0011700692,0.0011859747,0.00046966752],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004404786,0.0006878882,0.00045415637,0.0007831865,0.00037961808,0.0015601313,0.0013210683,0.00085619965,0.0047153695],"category_scores_gemma":[0.01674168,0.00046031928,0.0009902819,0.00056000095,0.00092883833,0.002969804,0.0013711101,0.0015247791,0.0006371159],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002927707,0.0009887195,0.0049391165,0.00071731466,0.000121186145,0.00008896711,0.0014678838,0.75591445,0.0065985043,0.10360258,0.0022569124,0.12037671],"study_design_scores_gemma":[0.00011542983,0.00048691666,0.00048108702,0.00003784948,0.00002119716,0.00002275019,0.00016918918,0.97930604,0.001049176,0.016344357,0.0019446637,0.00002149375],"about_ca_topic_score_codex":0.0050888075,"about_ca_topic_score_gemma":0.005179902,"teacher_disagreement_score":0.0050888075,"about_ca_system_score_codex":0.0016472137,"about_ca_system_score_gemma":0.0015464571,"threshold_uncertainty_score":0.023294985},"labels":[],"label_agreement":null},{"id":"W3095252018","doi":"10.1109/tse.2021.3070549","title":"Reinforcement Learning for Test Case Prioritization","year":2021,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Queen's University; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Reinforcement learning; Computer science; Prioritization; Context (archaeology); Ranking (information retrieval); Regression testing; Machine learning; Test case; Bidding; Test (biology); Adaptation (eye); Artificial intelligence; Regression analysis; Engineering; Software; Software system","score_opus":0.019405390937673923,"score_gpt":0.25133358157887337,"score_spread":0.23192819064119943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3095252018","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044530284,0.0005961345,0.950491,0.00042248267,0.000050715906,0.0002295436,0.00005841336,0.0014120727,0.002209305],"genre_scores_gemma":[0.8594105,0.0002341544,0.1380757,0.00023519936,0.00005395097,0.00029360692,0.00016989782,0.00011648532,0.0014103882],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963223,0.0017049817,0.00019311353,0.000692643,0.00071792403,0.00036898546],"domain_scores_gemma":[0.98490983,0.011297194,0.001338123,0.00076456415,0.0010951532,0.0005950878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050872313,0.0016333914,0.0013696098,0.0011302733,0.00045552646,0.0012069248,0.0021645795,0.001032445,0.0025040274],"category_scores_gemma":[0.021886146,0.00067581545,0.00066953036,0.00064972515,0.0013361719,0.0015663966,0.0012729962,0.0027242738,0.00047004188],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001974176,0.00029238602,0.00279596,0.00017980649,0.00008751017,0.00008833702,0.00011635881,0.8632915,0.0027415934,0.005763075,0.0011566782,0.123289436],"study_design_scores_gemma":[0.000024587447,0.00007101423,0.00020471972,0.000010811644,0.0000135920045,0.000016023605,0.000008640164,0.99583274,0.00065345987,0.0028764207,0.00028149126,0.0000065239597],"about_ca_topic_score_codex":0.0055242903,"about_ca_topic_score_gemma":0.0058353418,"teacher_disagreement_score":0.0055242903,"about_ca_system_score_codex":0.0017969427,"about_ca_system_score_gemma":0.0025833086,"threshold_uncertainty_score":0.026904166},"labels":[],"label_agreement":null},{"id":"W3111275476","doi":"10.1109/tse.2021.3078384","title":"A Comparison of Natural Language Understanding Platforms for Chatbots in Software Engineering","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":106,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Chatbot; Computer science; IBM; Natural language understanding; Natural language; Task (project management); Software; Data science; Artificial intelligence; Software engineering; World Wide Web; Natural language processing; Programming language; Engineering","score_opus":0.0374386258312366,"score_gpt":0.29901288892110517,"score_spread":0.2615742630898686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3111275476","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8653842,0.0070632547,0.06861119,0.0016520058,0.00049672084,0.0012978755,0.006637066,0.033044666,0.01581299],"genre_scores_gemma":[0.8310165,0.0018914195,0.11992458,0.0006229556,0.0001803138,0.0012731258,0.03761604,0.0016653218,0.0058097234],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9900748,0.0045545474,0.0010572454,0.0019354111,0.0019356225,0.000442288],"domain_scores_gemma":[0.9607784,0.029468695,0.0016697082,0.0024487711,0.0039311126,0.0017033222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010499058,0.0015912078,0.0011830279,0.0059866006,0.0011409379,0.002983351,0.0015977412,0.0022929572,0.0021445446],"category_scores_gemma":[0.03641912,0.0005395221,0.0013364104,0.0020345894,0.00090884685,0.00806092,0.003560568,0.002228377,0.002508084],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010056529,0.0030934208,0.052159704,0.007643786,0.0009171618,0.0012050521,0.01802421,0.021246303,0.049362037,0.0088897655,0.044920523,0.7824815],"study_design_scores_gemma":[0.0015009624,0.008492857,0.2226238,0.0019813941,0.00093008636,0.0023382357,0.017687254,0.5328187,0.055389166,0.017167458,0.13794082,0.0011292928],"about_ca_topic_score_codex":0.00555375,"about_ca_topic_score_gemma":0.008249082,"teacher_disagreement_score":0.010499058,"about_ca_system_score_codex":0.0016605905,"about_ca_system_score_gemma":0.0016728784,"threshold_uncertainty_score":0.055525005},"labels":[],"label_agreement":null},{"id":"W3114491449","doi":"10.1109/tse.2020.3045914","title":"Revisiting Test Impact Analysis in Continuous Testing From the Perspective of Code Dependencies","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Test Management Approach; Code coverage; Test case; Test (biology); Test harness; Source code; Software quality; Test suite; Regression testing; Code (set theory); Programming language; Reliability engineering; Software development; Software; Software construction; Machine learning; Set (abstract data type); Engineering","score_opus":0.02543541439253849,"score_gpt":0.26515398316104255,"score_spread":0.23971856876850406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3114491449","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6603282,0.0028947373,0.3265298,0.0015981556,0.00012817077,0.00033634907,0.00043527325,0.0019109624,0.0058384305],"genre_scores_gemma":[0.941621,0.00021327096,0.05730719,0.00013139355,0.00007216532,0.00009355217,0.00021571352,0.00014108293,0.00020451743],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9694372,0.015272592,0.0014975171,0.002757887,0.009997563,0.0010371648],"domain_scores_gemma":[0.6016458,0.34249964,0.022385938,0.016338782,0.015303763,0.001826053],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01974715,0.0016466517,0.0011040322,0.0063574617,0.0006431938,0.0026745116,0.0030674709,0.0010449657,0.0009992698],"category_scores_gemma":[0.16359049,0.00062189286,0.0009462982,0.004022824,0.0027228596,0.005091826,0.0018592107,0.0028878273,0.00023406083],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008743349,0.0015249122,0.29402488,0.0012881279,0.0006564542,0.0012508607,0.0023570913,0.17318743,0.01970174,0.016200505,0.0030726676,0.485861],"study_design_scores_gemma":[0.00014014651,0.0012286458,0.13822345,0.00047375294,0.00035213178,0.0007906722,0.0011108738,0.81829435,0.0126286,0.023129215,0.003467168,0.00016101378],"about_ca_topic_score_codex":0.0080089215,"about_ca_topic_score_gemma":0.0062582106,"teacher_disagreement_score":0.01974715,"about_ca_system_score_codex":0.0021607925,"about_ca_system_score_gemma":0.0025178327,"threshold_uncertainty_score":0.10443419},"labels":[],"label_agreement":null},{"id":"W3118614203","doi":"10.1109/tse.2020.3048335","title":"Accelerating Continuous Integration by Caching Environments and Inferring Dependencies","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; McGill University","funders":"Mitacs","keywords":"Computer science; Acceleration; Service (business); Dependency (UML); Task (project management); Distributed computing; Process (computing); Software; Software engineering; Operating system; Systems engineering","score_opus":0.018451724986903898,"score_gpt":0.2205484944750087,"score_spread":0.20209676948810482,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3118614203","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.48712516,0.002816434,0.36979514,0.001197232,0.00014945882,0.00038634503,0.027522596,0.10310407,0.007903511],"genre_scores_gemma":[0.49232423,0.0007277731,0.42801517,0.00025405336,0.00005253864,0.00027565742,0.07251174,0.0038227926,0.0020161318],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9952226,0.00080126507,0.0004635814,0.0013656899,0.0017858652,0.000361008],"domain_scores_gemma":[0.985,0.0051060785,0.0017118364,0.0056002075,0.0021833773,0.0003984941],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029848802,0.0018262967,0.00084282097,0.005907256,0.0008571355,0.002557745,0.0030265686,0.00085392967,0.0010374404],"category_scores_gemma":[0.01995222,0.0017468152,0.0014439244,0.0060351263,0.0010651865,0.004701651,0.0030815701,0.0020390917,0.0013993818],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008130749,0.00062250334,0.26221693,0.0012191227,0.0006241226,0.0008492937,0.0028962488,0.1735685,0.028376367,0.010871323,0.046245087,0.47169745],"study_design_scores_gemma":[0.00008259677,0.00022638176,0.04543383,0.00014756293,0.00031789686,0.0004133298,0.00063339924,0.8684969,0.031470165,0.014163469,0.038451366,0.00016309803],"about_ca_topic_score_codex":0.019033477,"about_ca_topic_score_gemma":0.04826032,"teacher_disagreement_score":0.019033477,"about_ca_system_score_codex":0.0011417118,"about_ca_system_score_gemma":0.002772157,"threshold_uncertainty_score":0.037845433},"labels":[],"label_agreement":null},{"id":"W3119230199","doi":"10.1109/tse.2021.3107680","title":"Mutation Analysis for Cyber-Physical Systems: Scalable Solutions and Results in the Space Domain","year":2021,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"H2020 European Research Council; Natural Sciences and Engineering Research Council of Canada; European Commission; European Space Agency","keywords":"Computer science; Context (archaeology); Software construction; Software; Software system; Scalability; Software reliability testing; Software engineering; Verification and validation; Avionics software; Domain (mathematical analysis); Reliability engineering; Engineering; Operating system","score_opus":0.02396113930530181,"score_gpt":0.25378680370064594,"score_spread":0.22982566439534413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119230199","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33464003,0.012757143,0.61867094,0.0051562344,0.0004506228,0.00040547756,0.00037197678,0.0041952203,0.02335239],"genre_scores_gemma":[0.69123286,0.00522077,0.29603848,0.0005420226,0.00024717324,0.0002792966,0.00059174077,0.0005546261,0.0052929507],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979359,0.0007822201,0.000070536094,0.00023739711,0.00073950033,0.0002344591],"domain_scores_gemma":[0.9876236,0.009811103,0.00037879046,0.0007904872,0.0009967806,0.0003993597],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037042864,0.0013080294,0.0013135675,0.0013923435,0.0005891883,0.0014108898,0.0012627202,0.0016591485,0.0031096896],"category_scores_gemma":[0.014430895,0.00026548607,0.00161687,0.0013828692,0.0016272615,0.0028933615,0.0018324082,0.0025743283,0.0004911979],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069382525,0.0010990166,0.0040915534,0.00069023034,0.00022322788,0.00055123144,0.00025554313,0.6278932,0.015396129,0.06251852,0.007388059,0.27919942],"study_design_scores_gemma":[0.00010329026,0.0001962132,0.0009554551,0.000050805258,0.00004702004,0.00007508358,0.00008325354,0.9665072,0.0049331854,0.025325196,0.0017030936,0.00002018946],"about_ca_topic_score_codex":0.0050728964,"about_ca_topic_score_gemma":0.0029388098,"teacher_disagreement_score":0.0050728964,"about_ca_system_score_codex":0.0013824352,"about_ca_system_score_gemma":0.0013048378,"threshold_uncertainty_score":0.019590318},"labels":[],"label_agreement":null},{"id":"W3120403074","doi":"10.1109/tse.2021.3101818","title":"Combining Genetic Programming and Model Checking to Generate Environment Assumptions","year":2021,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Evolutionary Algorithms and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; H2020 Excellent Science; Fonds National de la Recherche Luxembourg","keywords":"Component (thermodynamics); Computer science; Spurious relationship; Soundness; Flexibility (engineering); Benchmark (surveying); Genetic programming; Software; State (computer science); Artificial intelligence; Machine learning; Algorithm; Programming language; Mathematics","score_opus":0.01687721823519412,"score_gpt":0.22193729267168105,"score_spread":0.20506007443648694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3120403074","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020988703,0.00008647426,0.97438025,0.00020925327,0.000027598115,0.00008068539,0.0001391782,0.0027134875,0.0013744638],"genre_scores_gemma":[0.42461088,0.0001958412,0.5721382,0.0002605287,0.000036118276,0.00021520785,0.0007393831,0.00060387974,0.0011998488],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99709094,0.0010686093,0.00013032972,0.0006195386,0.00087255373,0.00021800408],"domain_scores_gemma":[0.9830676,0.013611845,0.0008332565,0.0015812219,0.00080197334,0.000104143604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030858328,0.0021084102,0.0010011155,0.002183291,0.0006793241,0.0014814015,0.0023007132,0.0017462706,0.0018318049],"category_scores_gemma":[0.018951312,0.001030596,0.002130872,0.0010848509,0.0017782366,0.002417619,0.0018957811,0.0027336383,0.00049912813],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054552813,0.00008554236,0.002914682,0.000105776984,0.00007616629,0.00015791583,0.000080526705,0.9324885,0.0019903672,0.010902705,0.00066110684,0.050482143],"study_design_scores_gemma":[0.0000075507373,0.000014475774,0.0000964122,0.000013839278,0.000013262907,0.000019071775,0.000009608026,0.9882931,0.0013802708,0.00975773,0.0003887912,0.0000059460203],"about_ca_topic_score_codex":0.009372694,"about_ca_topic_score_gemma":0.015404188,"teacher_disagreement_score":0.009372694,"about_ca_system_score_codex":0.0014375822,"about_ca_system_score_gemma":0.002740358,"threshold_uncertainty_score":0.018636227},"labels":[],"label_agreement":null},{"id":"W3127656560","doi":"10.1109/tse.2021.3055123","title":"Deprecation of Packages and Releases in Software Ecosystems: A Case Study on NPM","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Notation; Software; Code (set theory); Programming language; Software engineering; Information retrieval; World Wide Web; Arithmetic; Mathematics; Set (abstract data type)","score_opus":0.01918480495773581,"score_gpt":0.2594253268007716,"score_spread":0.24024052184303582,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3127656560","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9888692,0.00033993102,0.0044751046,0.0010546488,0.000019541672,0.00006891584,0.00009180558,0.00014334216,0.004937492],"genre_scores_gemma":[0.9869372,0.00033417562,0.009584881,0.00030396206,0.000033710232,0.000084987645,0.00019971652,0.00010708912,0.0024141206],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9937516,0.0031827455,0.00041505918,0.0006569414,0.001409376,0.000584203],"domain_scores_gemma":[0.95123416,0.03320478,0.00726484,0.0039366516,0.0025221591,0.0018373859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007026562,0.00036306636,0.00030204875,0.00227705,0.003939587,0.0023092753,0.0013859707,0.002055843,0.0015213796],"category_scores_gemma":[0.029985579,0.00040909316,0.0005817636,0.002847458,0.002733676,0.005051306,0.0029667756,0.0018716092,0.00044178448],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005279112,0.0019043346,0.5112999,0.0008824654,0.00010859475,0.056217916,0.21023716,0.0061808433,0.010097101,0.025628503,0.0108716665,0.16604364],"study_design_scores_gemma":[0.00013327113,0.0010506245,0.5102687,0.0009857701,0.00017439065,0.028574906,0.2378517,0.041041423,0.012131937,0.014304724,0.15317412,0.00030839708],"about_ca_topic_score_codex":0.009061076,"about_ca_topic_score_gemma":0.0141529655,"teacher_disagreement_score":0.009061076,"about_ca_system_score_codex":0.0024953648,"about_ca_system_score_gemma":0.0021182187,"threshold_uncertainty_score":0.037160516},"labels":[],"label_agreement":null},{"id":"W3132847876","doi":"10.1109/tse.2021.3060918","title":"Studying Duplicate Logging Statements and Their Relationships With Code Clones","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Logging; Code (set theory); Programming language; Database; Software engineering; Ecology","score_opus":0.040146525901182346,"score_gpt":0.26808989881282014,"score_spread":0.2279433729116378,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3132847876","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.978735,0.001593537,0.015659062,0.00035713505,0.000039693783,0.00012756808,0.0005733741,0.0015626549,0.0013519215],"genre_scores_gemma":[0.9796464,0.00037735258,0.01726821,0.00015319706,0.00003636772,0.00008457329,0.001291452,0.00034240916,0.0008001925],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.985642,0.0022621385,0.0015284661,0.003847396,0.006058429,0.00066161255],"domain_scores_gemma":[0.71903384,0.16245653,0.073515505,0.017837621,0.02503091,0.002125563],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068012765,0.0007919438,0.00056082755,0.006175816,0.0011506375,0.0014385126,0.0013360115,0.00119314,0.00085340213],"category_scores_gemma":[0.121963635,0.0006527671,0.00051418116,0.0043501677,0.0015931057,0.0033079986,0.0017347571,0.0014381203,0.00027223324],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001644948,0.00013768785,0.91564775,0.0005191847,0.00015832488,0.0025276488,0.0043940227,0.0019863038,0.005736636,0.0008434416,0.0018767336,0.06600791],"study_design_scores_gemma":[0.00006600395,0.0004777842,0.86952335,0.00056637963,0.00073232123,0.013273334,0.00652915,0.059672784,0.027502475,0.00473963,0.016708383,0.00020840047],"about_ca_topic_score_codex":0.0055264076,"about_ca_topic_score_gemma":0.009856518,"teacher_disagreement_score":0.0068012765,"about_ca_system_score_codex":0.0010048135,"about_ca_system_score_gemma":0.0015381376,"threshold_uncertainty_score":0.03596902},"labels":[],"label_agreement":null},{"id":"W3133304533","doi":"10.1109/tse.2021.3058985","title":"A Study of C/C++ Code Weaknesses on Stack Overflow","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Polytechnique Montréal; University of Manitoba; Concordia University; Huawei Technologies (Canada)","funders":"","keywords":"Notation; Computer science; Code (set theory); Stack (abstract data type); Programming language; Mathematical notation; Software; Theoretical computer science; Mathematics; Arithmetic","score_opus":0.02446365343770417,"score_gpt":0.26541750701339983,"score_spread":0.24095385357569565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3133304533","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9922402,0.0005698143,0.0029370307,0.00045204518,0.000017837297,0.000056199133,0.0007367576,0.0002998415,0.0026902852],"genre_scores_gemma":[0.9900119,0.000336239,0.005873804,0.00021560008,0.00003169704,0.00006389046,0.0017824121,0.00024317605,0.0014412568],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9954987,0.00092498004,0.00040661974,0.00096853345,0.0019182526,0.0002829363],"domain_scores_gemma":[0.890596,0.07029823,0.023795426,0.004198512,0.009617789,0.0014940938],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036640675,0.00064916775,0.00034769782,0.0051942198,0.0011014139,0.0015510353,0.00081411336,0.001386025,0.0019443635],"category_scores_gemma":[0.07751864,0.0003987584,0.00037578613,0.0036934654,0.0016212395,0.0047479793,0.0018592627,0.0013813666,0.00069284503],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043487293,0.0002294251,0.8528728,0.000893318,0.00013734985,0.0017544654,0.01693768,0.003099532,0.0076708524,0.0033475088,0.00724287,0.105379365],"study_design_scores_gemma":[0.000035016732,0.0004818621,0.87544817,0.0008333297,0.00019903976,0.0063511175,0.017257322,0.053402625,0.013318527,0.005326613,0.02713246,0.00021393882],"about_ca_topic_score_codex":0.005307372,"about_ca_topic_score_gemma":0.006695543,"teacher_disagreement_score":0.005307372,"about_ca_system_score_codex":0.0007584026,"about_ca_system_score_gemma":0.00087954383,"threshold_uncertainty_score":0.019377649},"labels":[],"label_agreement":null},{"id":"W3134221463","doi":"10.1109/tse.2021.3064953","title":"Uncovering the Benefits and Challenges of Continuous Integration Practices","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Context (archaeology); Best practice; Suite; Process (computing); Software; Process management; Quality (philosophy); Knowledge management; Software development; Software development process; Data science; Software engineering; Business; Management","score_opus":0.032604655657231134,"score_gpt":0.25456562614295214,"score_spread":0.221960970485721,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3134221463","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9325812,0.0012859943,0.03177114,0.01457691,0.00009070766,0.00026302788,0.00004054222,0.0001675348,0.019223047],"genre_scores_gemma":[0.9866556,0.00040414307,0.011694056,0.00037416857,0.000013641103,0.00011199703,0.000026294692,0.000031002048,0.0006891247],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.943271,0.03097667,0.0028618372,0.0046804436,0.014735676,0.0034743033],"domain_scores_gemma":[0.88889927,0.0719487,0.0099041,0.011604226,0.013820383,0.0038232862],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05096928,0.0007649219,0.00052309164,0.003477715,0.00649462,0.012445914,0.0031419576,0.0025346857,0.0008353179],"category_scores_gemma":[0.09074743,0.0011813365,0.00047372153,0.0039870464,0.010576969,0.01504756,0.010229914,0.0053633596,0.00017910394],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001040099,0.00040378663,0.100493,0.00054342847,0.00007797191,0.0017364329,0.67422026,0.0011712548,0.0039509195,0.044321902,0.0022145864,0.17076242],"study_design_scores_gemma":[0.00008082496,0.0007499618,0.06627843,0.0017058548,0.00012812286,0.0016352,0.81095135,0.0102608325,0.003855061,0.047125358,0.057082944,0.00014608234],"about_ca_topic_score_codex":0.007836853,"about_ca_topic_score_gemma":0.011154512,"teacher_disagreement_score":0.05096928,"about_ca_system_score_codex":0.0092922,"about_ca_system_score_gemma":0.014356362,"threshold_uncertainty_score":0.26955456},"labels":[],"label_agreement":null},{"id":"W3136085434","doi":"10.1109/tse.2021.3066330","title":"Continuously Managing NFRs: Opportunities and Challenges in Practice","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Maintainability; Computer science; Agile software development; Software; Risk analysis (engineering); Software engineering; Process management; Engineering; Business","score_opus":0.040420160785735494,"score_gpt":0.2368191328294542,"score_spread":0.1963989720437187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3136085434","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51981807,0.008491406,0.16699268,0.21695955,0.00075213466,0.0007922718,0.00011062419,0.0009688934,0.085114315],"genre_scores_gemma":[0.92438084,0.0024888327,0.065516174,0.0042346893,0.00018478138,0.00044541023,0.000056839403,0.00015379224,0.0025385101],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.85881495,0.0927088,0.0064734323,0.010817558,0.021583823,0.009601562],"domain_scores_gemma":[0.73311406,0.18765071,0.018767286,0.021418326,0.024289733,0.014759924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08895434,0.0012524399,0.0010240908,0.0032184783,0.01073649,0.023124063,0.0065780636,0.009119995,0.004424896],"category_scores_gemma":[0.13904266,0.0013246981,0.0009332925,0.0032672954,0.018306652,0.03283311,0.013874767,0.008041213,0.0011997244],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019829215,0.002115641,0.057166547,0.0021122359,0.0001590001,0.004622058,0.34100777,0.005873839,0.0038038273,0.14260553,0.01564716,0.42468813],"study_design_scores_gemma":[0.00013740701,0.0009718925,0.012709574,0.00386254,0.000083464365,0.0030304764,0.5325154,0.014930154,0.0022314708,0.28462294,0.144566,0.00033870156],"about_ca_topic_score_codex":0.005038059,"about_ca_topic_score_gemma":0.008367952,"teacher_disagreement_score":0.08895434,"about_ca_system_score_codex":0.010058582,"about_ca_system_score_gemma":0.022325369,"threshold_uncertainty_score":0.4704411},"labels":[],"label_agreement":null},{"id":"W3144273096","doi":"10.1109/tse.2021.3068901","title":"On the Untriviality of Trivial Packages: An Empirical Study of npm JavaScript Packages","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Concordia University","funders":"","keywords":"Computer science; JavaScript; Programming language; Empirical research; Software engineering","score_opus":0.0341506444947318,"score_gpt":0.29139461396325345,"score_spread":0.2572439694685216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3144273096","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9963972,0.00018832619,0.0014847482,0.00018046118,0.0000056578815,0.000019920262,0.00019654445,0.00002873164,0.0014983782],"genre_scores_gemma":[0.9971624,0.0001347639,0.001618977,0.00007351218,0.000014850143,0.000032662138,0.00047175188,0.00006156699,0.00042946637],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9927089,0.0034006615,0.00041724424,0.0012520032,0.0017496251,0.00047151954],"domain_scores_gemma":[0.86486655,0.09856489,0.02085099,0.0050157434,0.0071316822,0.0035701983],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0087051485,0.00035556045,0.00044865717,0.003949329,0.0014463059,0.0022348643,0.001244349,0.0010294714,0.0019779492],"category_scores_gemma":[0.06867024,0.00038785604,0.0003605428,0.0040819743,0.00270992,0.006343861,0.0021090866,0.0016170955,0.000784976],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013891289,0.00024321358,0.943463,0.00030202328,0.00008365683,0.0005277758,0.023863818,0.00052909896,0.0012270822,0.0027136344,0.002616284,0.02429156],"study_design_scores_gemma":[0.000018344417,0.0002153545,0.9292747,0.00021874021,0.00006232598,0.001490836,0.042642456,0.011149407,0.0009153609,0.002690806,0.01126432,0.00005733596],"about_ca_topic_score_codex":0.0022061467,"about_ca_topic_score_gemma":0.003401791,"teacher_disagreement_score":0.0087051485,"about_ca_system_score_codex":0.0006191703,"about_ca_system_score_gemma":0.00051761576,"threshold_uncertainty_score":0.046037793},"labels":[],"label_agreement":null},{"id":"W3148779029","doi":"10.1109/tse.2020.2988396","title":"A3: Assisting Android API Migrations Using Code Examples","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Android (operating system); Computer science; Documentation; Application programming interface; Source code; World Wide Web; Java; Operating system","score_opus":0.051130970689089575,"score_gpt":0.26172621043122185,"score_spread":0.2105952397421323,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3148779029","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5546994,0.0012483133,0.2816092,0.0018355617,0.00038488398,0.0014112804,0.00393692,0.14308988,0.011784452],"genre_scores_gemma":[0.4817454,0.00029964492,0.5034999,0.00040914243,0.000037165068,0.00050250086,0.005353562,0.0011457752,0.007006929],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988985,0.0002829865,0.00008216212,0.00036856902,0.00030746558,0.000060343365],"domain_scores_gemma":[0.99433446,0.0027524303,0.00055652176,0.00084791373,0.0012266,0.0002820887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012113035,0.0015930943,0.0005193779,0.0013590202,0.00048592917,0.00091277284,0.0020955626,0.0014810023,0.0028282069],"category_scores_gemma":[0.014230021,0.00052816793,0.00062914996,0.00059339503,0.0003572801,0.0017841535,0.0011028511,0.0012501094,0.0018468774],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00087762723,0.0020416027,0.037965544,0.0010266372,0.00018982602,0.0008906446,0.0017642888,0.04702195,0.02868453,0.0016409995,0.04380671,0.83408964],"study_design_scores_gemma":[0.00015429554,0.00046286837,0.0076549305,0.00013179424,0.00008792342,0.00041525785,0.00042541773,0.9450511,0.02426322,0.0019700648,0.019317074,0.000066072775],"about_ca_topic_score_codex":0.007638099,"about_ca_topic_score_gemma":0.014184008,"teacher_disagreement_score":0.007638099,"about_ca_system_score_codex":0.00046760633,"about_ca_system_score_gemma":0.0013463187,"threshold_uncertainty_score":0.0151872635},"labels":[],"label_agreement":null},{"id":"W3152301934","doi":"10.1109/tse.2021.3070269","title":"Software Batch Testing to Save Build Test Resources and to Reduce Feedback Time","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Commit; Test case; Test (biology); Reliability engineering; Database; Machine learning","score_opus":0.015497159235501757,"score_gpt":0.23784210009981144,"score_spread":0.22234494086430967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3152301934","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2635239,0.0023043354,0.6229098,0.0012149945,0.00053524104,0.0014121749,0.001570835,0.08558472,0.02094407],"genre_scores_gemma":[0.66028893,0.00037598272,0.32423618,0.00045386355,0.00009240934,0.000651737,0.0021870537,0.0049784207,0.0067354306],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.991063,0.0022846307,0.0005890718,0.0015702965,0.0037030736,0.0007899482],"domain_scores_gemma":[0.9482468,0.024068533,0.0031782947,0.015726913,0.0070002754,0.001779087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008526418,0.0021493153,0.0011871787,0.0022182495,0.00065137004,0.0022167924,0.004647088,0.000997464,0.009535929],"category_scores_gemma":[0.03448448,0.0012295317,0.001192333,0.0017093591,0.0012839638,0.004589143,0.0019735969,0.0030600517,0.0030828773],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003085216,0.001591555,0.030173708,0.0014886972,0.00045448256,0.00048665857,0.00094231835,0.19172874,0.079805136,0.015041976,0.03126182,0.64393973],"study_design_scores_gemma":[0.0008402917,0.00294765,0.019349461,0.0002996165,0.0003420017,0.00070092786,0.00044538843,0.8174607,0.09863373,0.020700391,0.03797436,0.00030547334],"about_ca_topic_score_codex":0.008485492,"about_ca_topic_score_gemma":0.009800539,"teacher_disagreement_score":0.009535929,"about_ca_system_score_codex":0.0018998402,"about_ca_system_score_gemma":0.0035903822,"threshold_uncertainty_score":0.045092523},"labels":[],"label_agreement":null},{"id":"W3153182107","doi":"10.1109/tse.2021.3073773","title":"On the Relationship Between the Developer’s Perceptible Race and Ethnicity and the Evaluation of Contributions in OSS","year":2021,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Ethnic group; Race (biology); Diversity (politics); White (mutation); Empirical research; Computer science; Open source software; Software; Data science; Sociology; Mathematics; Gender studies; Statistics","score_opus":0.04667324933384913,"score_gpt":0.28521615133311873,"score_spread":0.23854290199926959,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3153182107","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9916892,0.00012231205,0.0019890522,0.00037613197,0.000032779244,0.00013360957,0.00011475851,0.000008009277,0.005534155],"genre_scores_gemma":[0.9952324,0.00010210307,0.0019851278,0.00015715197,0.000025349695,0.00039941206,0.0001608137,0.00001525698,0.0019223463],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98056406,0.012709078,0.0013654069,0.0017371662,0.002477838,0.0011463925],"domain_scores_gemma":[0.67026794,0.26638928,0.032248225,0.008981235,0.01704319,0.0050702067],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03364614,0.0005169735,0.00049833255,0.0024495346,0.0020878094,0.0031575973,0.00096204435,0.0012644588,0.0074626603],"category_scores_gemma":[0.11467705,0.00038130808,0.0011507061,0.0023892806,0.0028273626,0.0025270213,0.003485383,0.0018377221,0.000922988],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030347385,0.0006207759,0.9671214,0.00009213253,0.00012540413,0.000093001734,0.019041296,0.00034461712,0.00033559438,0.0012193008,0.00047155868,0.010231504],"study_design_scores_gemma":[0.00002452569,0.00044362398,0.96844804,0.00018879081,0.00018515339,0.00006969171,0.024392689,0.0017113417,0.00097629585,0.0012555317,0.0022575583,0.000046759047],"about_ca_topic_score_codex":0.013195021,"about_ca_topic_score_gemma":0.019855184,"teacher_disagreement_score":0.96635383,"about_ca_system_score_codex":0.0025278414,"about_ca_system_score_gemma":0.0041659554,"threshold_uncertainty_score":0.17793989},"labels":[],"label_agreement":null},{"id":"W3156363234","doi":"10.1109/tse.2020.3023955","title":"PerfJIT: Test-Level Just-in-Time Prediction for Performance Regression Introducing Commits","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Commit; Computer science; Regression; Machine learning; Regression analysis; Regression testing; Test (biology); Artificial intelligence; Performance prediction; Linear regression; Code (set theory); Data mining; Statistics; Database; Software; Set (abstract data type); Programming language; Software system; Mathematics","score_opus":0.02316888172101865,"score_gpt":0.22436103631883705,"score_spread":0.20119215459781842,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3156363234","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3417269,0.0012055081,0.36669964,0.0008697502,0.0006037353,0.0006520888,0.01538744,0.2688992,0.0039556995],"genre_scores_gemma":[0.78728443,0.00025467123,0.17748058,0.0003471813,0.0001955655,0.00061088556,0.025998842,0.0044952985,0.0033325225],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968982,0.0005153187,0.0002765316,0.0009832989,0.0010114608,0.00031518497],"domain_scores_gemma":[0.98396957,0.007951256,0.0020071627,0.002910506,0.0023420297,0.00081950193],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032476594,0.0029218907,0.0010403304,0.0023246673,0.00043236,0.0012572471,0.00255774,0.0011363751,0.0027744635],"category_scores_gemma":[0.02649946,0.0007921076,0.0009578317,0.001357554,0.00053375366,0.0024375545,0.0009854955,0.002805515,0.0023349726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015400008,0.0013442442,0.20884927,0.00078329246,0.0003948891,0.0006103593,0.000377143,0.2222656,0.022719217,0.0019101299,0.07942494,0.45978087],"study_design_scores_gemma":[0.000041200263,0.00028343397,0.0124359345,0.00002146927,0.00003204273,0.00012258474,0.000026493823,0.97401285,0.008183488,0.0016260443,0.003176648,0.000037758196],"about_ca_topic_score_codex":0.008273213,"about_ca_topic_score_gemma":0.0101340795,"teacher_disagreement_score":0.008273213,"about_ca_system_score_codex":0.0008266317,"about_ca_system_score_gemma":0.001868922,"threshold_uncertainty_score":0.017175496},"labels":[],"label_agreement":null},{"id":"W3162606549","doi":"10.1109/tse.2021.3081171","title":"Context-Aware Personalized Crowdtesting Task Recommendation","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Task (project management); Context (archaeology); Human–computer interaction; World Wide Web; Data science; Software engineering; Systems engineering","score_opus":0.013921702004545542,"score_gpt":0.21737926802816468,"score_spread":0.20345756602361914,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3162606549","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22461712,0.0023712164,0.7380632,0.0012953824,0.00031739473,0.0013521882,0.0025465703,0.012726927,0.016709903],"genre_scores_gemma":[0.69213593,0.0006548278,0.29714918,0.00032853096,0.00014359795,0.0006213938,0.0022169615,0.00033063185,0.006418993],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984768,0.00041908261,0.000095667725,0.00048719838,0.000359254,0.00016204712],"domain_scores_gemma":[0.9970822,0.0011330174,0.00022926161,0.00056795037,0.00071041577,0.00027715034],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014578591,0.0014178684,0.0013725028,0.0018951406,0.0010198171,0.0012218974,0.00198185,0.0011639638,0.0030628259],"category_scores_gemma":[0.005978284,0.00059750275,0.0008099909,0.0012963549,0.0003301905,0.0016674673,0.0013288992,0.0010524888,0.0018920781],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010993002,0.00142774,0.03874372,0.0007311868,0.00036821896,0.0005128083,0.0023904124,0.0922984,0.026282258,0.0035953915,0.038215134,0.7943354],"study_design_scores_gemma":[0.00014738788,0.00044246696,0.018787788,0.00014616198,0.00026161814,0.00035865826,0.0014933673,0.9148932,0.015364968,0.0090986565,0.03881793,0.00018778222],"about_ca_topic_score_codex":0.014914054,"about_ca_topic_score_gemma":0.03237931,"teacher_disagreement_score":0.014914054,"about_ca_system_score_codex":0.0007080329,"about_ca_system_score_gemma":0.0014274605,"threshold_uncertainty_score":0.029654503},"labels":[],"label_agreement":null},{"id":"W3163202066","doi":"10.1109/tse.2021.3082068","title":"An Empirical Study of Type-Related Defects in Python Projects","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Python (programming language); Computer science; Programming language; Artificial intelligence","score_opus":0.025406470675922346,"score_gpt":0.2945407086543776,"score_spread":0.26913423797845526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3163202066","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.994288,0.00023661212,0.0018440222,0.00040152043,0.000023698847,0.00019949398,0.0004966014,0.00009081378,0.0024192785],"genre_scores_gemma":[0.99571496,0.00015844184,0.001933995,0.00014662085,0.000019312201,0.0003808391,0.00067821564,0.0000660596,0.00090146635],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.956293,0.016468026,0.0051749256,0.0049427026,0.01534587,0.0017755631],"domain_scores_gemma":[0.43530193,0.31167158,0.15736502,0.019454086,0.06721427,0.008993046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.034676667,0.00047196238,0.00051609246,0.0057401387,0.0017033606,0.0029589788,0.0018746298,0.0018389908,0.002532884],"category_scores_gemma":[0.318782,0.00066147454,0.00052884495,0.005409111,0.002998511,0.0059979553,0.0033795324,0.0030026897,0.00090512395],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002706387,0.00061424705,0.9561905,0.00028846512,0.0000630402,0.00039988465,0.013401199,0.00034617263,0.0003946144,0.00086054014,0.0023308946,0.024839826],"study_design_scores_gemma":[0.000060637103,0.0010190569,0.9527838,0.00041370618,0.00008172021,0.0010758527,0.028973626,0.0039531062,0.0012977354,0.0012946129,0.008951855,0.00009424749],"about_ca_topic_score_codex":0.0038519094,"about_ca_topic_score_gemma":0.0042889123,"teacher_disagreement_score":0.034676667,"about_ca_system_score_codex":0.0025440012,"about_ca_system_score_gemma":0.0029715213,"threshold_uncertainty_score":0.18338996},"labels":[],"label_agreement":null},{"id":"W3167387776","doi":"10.1109/tse.2021.3083715","title":"LogAssist: Assisting Log Analysis Through Log Summarization","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Concordia University","funders":"","keywords":"Computer science; Automatic summarization; Workflow; Debugging; Event (particle physics); Information retrieval; Data mining; Programming language; Database","score_opus":0.013454505144419968,"score_gpt":0.23017273441904804,"score_spread":0.21671822927462808,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3167387776","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046141896,0.00078261894,0.60820395,0.0010273419,0.00022138513,0.00093084516,0.010559725,0.32947698,0.0026551953],"genre_scores_gemma":[0.2148126,0.0006550466,0.7461563,0.0004580998,0.00015867343,0.00062857906,0.028515335,0.0042492403,0.004366128],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979886,0.00047862934,0.0002387711,0.0003674268,0.00081233704,0.00011427293],"domain_scores_gemma":[0.9934202,0.0033455964,0.00064704136,0.0012977397,0.00096259906,0.0003268881],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022140183,0.0028663832,0.0010303633,0.0041573197,0.0007339434,0.0020842487,0.002932048,0.0009834446,0.004332127],"category_scores_gemma":[0.011672533,0.00076764345,0.00089823804,0.0022831692,0.0005178647,0.005182952,0.0020602597,0.0018202412,0.0024866431],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013443325,0.0012498416,0.015403384,0.0013916601,0.00030966298,0.00069614727,0.0012040529,0.036572926,0.03072078,0.004483015,0.08838215,0.8182421],"study_design_scores_gemma":[0.00020245509,0.0003910495,0.006132999,0.000119166434,0.00011247075,0.0004082936,0.0005608627,0.9029347,0.039524954,0.009261687,0.04014271,0.0002086593],"about_ca_topic_score_codex":0.009683784,"about_ca_topic_score_gemma":0.013796195,"teacher_disagreement_score":0.009683784,"about_ca_system_score_codex":0.00065597636,"about_ca_system_score_gemma":0.0024349447,"threshold_uncertainty_score":0.019254863},"labels":[],"label_agreement":null},{"id":"W3170417470","doi":"10.1109/tse.2021.3124332","title":"AI-Enabled Automation for Completeness Checking of Privacy Policies","year":2021,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Privacy, Security, and Data Protection","field":"Social Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Privacy policy; Completeness (order theory); General Data Protection Regulation; Information privacy; Privacy software; Privacy by Design; Privacy law; Computer security; Data Protection Act 1998","score_opus":0.03746098878162252,"score_gpt":0.3041413051598377,"score_spread":0.2666803163782152,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3170417470","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013890071,0.00012587995,0.97838163,0.00071151886,0.000032483662,0.00041997415,0.00040593024,0.0043575945,0.0016749519],"genre_scores_gemma":[0.19164279,0.00016248989,0.80504876,0.00028455784,0.000026667505,0.00048919575,0.0012982856,0.00033652957,0.0007107477],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.95526415,0.02270462,0.004695989,0.0052158623,0.0109377615,0.0011816703],"domain_scores_gemma":[0.87994933,0.07725108,0.0076406114,0.021356419,0.013061783,0.0007408039],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017554311,0.0016111434,0.0012070299,0.005692413,0.0019270005,0.004294098,0.0028585764,0.0019255952,0.0025967048],"category_scores_gemma":[0.07721704,0.0013301844,0.0035341913,0.0025189596,0.004251233,0.0061195465,0.0054587442,0.0033287338,0.0010538885],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006017897,0.00083616306,0.018529234,0.003364474,0.0005546532,0.0014281584,0.00917013,0.14360347,0.06520543,0.19084017,0.008336755,0.55752957],"study_design_scores_gemma":[0.000112656366,0.00022328096,0.0031916325,0.00046699625,0.00017466307,0.00072420726,0.0015450425,0.71427065,0.061540615,0.19333568,0.024239194,0.00017543115],"about_ca_topic_score_codex":0.008952052,"about_ca_topic_score_gemma":0.008531895,"teacher_disagreement_score":0.017554311,"about_ca_system_score_codex":0.0033632594,"about_ca_system_score_gemma":0.0087311715,"threshold_uncertainty_score":0.092837214},"labels":[],"label_agreement":null},{"id":"W3173415420","doi":"10.1109/tse.2021.3092692","title":"An Experience Report on Producing Verifiable Builds for Large-Scale Commercial Systems","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Huawei Technologies (Canada)","funders":"","keywords":"Toolchain; Verifiable secret sharing; Computer science; Software engineering; Process (computing); TRACE (psycholinguistics); Code (set theory); Software; Notation; Programming language; Theoretical computer science; Set (abstract data type)","score_opus":0.019071393867009387,"score_gpt":0.28494894857246433,"score_spread":0.26587755470545493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173415420","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2376789,0.0024782773,0.68411684,0.002726533,0.00025393232,0.0013869121,0.0010642132,0.012747587,0.057546813],"genre_scores_gemma":[0.44596785,0.0014735148,0.52992415,0.00030162276,0.00005851516,0.00041269758,0.0020575908,0.003825562,0.01597851],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9851413,0.00546413,0.0011810216,0.0014083133,0.006077231,0.00072803244],"domain_scores_gemma":[0.9612722,0.014730188,0.0014849873,0.014400522,0.006781824,0.0013302692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018135313,0.00088287715,0.00040242556,0.0021387993,0.0016899502,0.003313747,0.0020267689,0.0012559072,0.008240258],"category_scores_gemma":[0.04091493,0.00065453845,0.00096226664,0.0017443508,0.0021735914,0.004469668,0.00501669,0.0018896968,0.0034955933],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034754683,0.00082362024,0.01598071,0.0015224084,0.0001306824,0.0035232939,0.039807245,0.017556218,0.055292822,0.033222932,0.02685644,0.80493605],"study_design_scores_gemma":[0.00019784411,0.00226253,0.02056559,0.0014093072,0.00027994736,0.0067463587,0.014766899,0.04805591,0.18325497,0.022714714,0.6993101,0.00043582052],"about_ca_topic_score_codex":0.0034644275,"about_ca_topic_score_gemma":0.004756299,"teacher_disagreement_score":0.018135313,"about_ca_system_score_codex":0.0014878928,"about_ca_system_score_gemma":0.0030321607,"threshold_uncertainty_score":0.095909834},"labels":[],"label_agreement":null},{"id":"W3187068675","doi":"10.1109/tse.2021.3099532","title":"What Makes Agile Software Development Agile?","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Calgary","funders":"European Regional Development Fund; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Christian Doppler Forschungsgesellschaft; Bundesministerium für Digitalisierung und Wirtschaftsstandort; Science Foundation Ireland; Ministerio de Ciencia, Innovación y Universidades; Österreichische Nationalstiftung für Forschung, Technologie und Entwicklung","keywords":"Agile software development; Extreme programming practices; Computer science; Agile Unified Process; Lean software development; Software development; Software; Agile usability engineering; Process management; Process (computing); Software development process; Empirical research; Software engineering; Knowledge management; Engineering management; Engineering","score_opus":0.016053542016305346,"score_gpt":0.23468772974138075,"score_spread":0.2186341877250754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3187068675","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08988881,0.07857501,0.10237516,0.5892243,0.013150276,0.00020401072,0.0003459392,0.0010059137,0.12523063],"genre_scores_gemma":[0.80060273,0.073955774,0.065926045,0.03747091,0.008339802,0.00039071147,0.00034016345,0.0006467807,0.012327268],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9803021,0.010493191,0.0009352416,0.0012134106,0.005829125,0.0012268889],"domain_scores_gemma":[0.94091594,0.036319524,0.0050934916,0.0041811382,0.0095746275,0.003915358],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014350291,0.00065006147,0.000603691,0.0017500418,0.0019453253,0.008115534,0.0008267483,0.0032968442,0.0027240408],"category_scores_gemma":[0.055664517,0.0005912886,0.00062374905,0.0021779558,0.0063550314,0.012388718,0.0029557887,0.0046133325,0.0011811615],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000109143424,0.00018997412,0.028900344,0.0017760199,0.00019187246,0.00085838296,0.024467845,0.0020905137,0.0021373811,0.1874952,0.07715918,0.67462415],"study_design_scores_gemma":[0.000059680988,0.00016042351,0.01827592,0.004002973,0.00011783549,0.001758117,0.029585045,0.002643119,0.0013201651,0.32573292,0.6162036,0.00014023836],"about_ca_topic_score_codex":0.0023539448,"about_ca_topic_score_gemma":0.0021353776,"teacher_disagreement_score":0.014350291,"about_ca_system_score_codex":0.001348307,"about_ca_system_score_gemma":0.0033193582,"threshold_uncertainty_score":0.07589251},"labels":[],"label_agreement":null},{"id":"W3196126762","doi":"10.1109/tse.2021.3106247","title":"Dependency Smells in JavaScript Projects","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code smell; Computer science; Dependency (UML); Commit; JavaScript; Code refactoring; Dependency graph; Software engineering; Security bug; World Wide Web; Popularity; Computer security; Data science; Software; Software development; Software quality; Software security assurance; Database; Programming language; Information security","score_opus":0.018662997445303465,"score_gpt":0.2381131029342694,"score_spread":0.21945010548896593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196126762","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9960983,0.00051651296,0.0014868863,0.00025794472,0.000008509169,0.00002781513,0.00078092044,0.00010687999,0.00071623264],"genre_scores_gemma":[0.99444276,0.00039085638,0.0023251106,0.00008296178,0.000020141419,0.00007472847,0.0021043916,0.00006604085,0.00049306307],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9888214,0.0030488337,0.0020746617,0.0016773775,0.0037677935,0.0006099677],"domain_scores_gemma":[0.7712713,0.109123446,0.09120293,0.0092389835,0.015140022,0.0040232986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074820286,0.00029626087,0.00034841752,0.007023984,0.00097427913,0.001586357,0.0006663695,0.00085343193,0.0007717127],"category_scores_gemma":[0.081374176,0.00044928587,0.0004488257,0.007600724,0.0009128446,0.0034782272,0.003124486,0.0012932043,0.00031354977],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009694454,0.000063779575,0.95849115,0.0002665282,0.00004612022,0.00048340842,0.009732634,0.00037511947,0.0012595925,0.00022711867,0.0016944885,0.02726311],"study_design_scores_gemma":[0.000006017791,0.00010358118,0.984209,0.0001699406,0.000023158926,0.0010725698,0.0074362475,0.0017140328,0.00074659433,0.00042817896,0.004045311,0.000045308738],"about_ca_topic_score_codex":0.0046733376,"about_ca_topic_score_gemma":0.009186126,"teacher_disagreement_score":0.0074820286,"about_ca_system_score_codex":0.00094832486,"about_ca_system_score_gemma":0.00085656927,"threshold_uncertainty_score":0.0395692},"labels":[],"label_agreement":null},{"id":"W3199338376","doi":"10.1109/tse.2021.3112503","title":"How Templated Requirements Specifications Inhibit Creativity in Software Engineering","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Formality; Creativity; Coding (social sciences); Software engineering; Requirements engineering; Software; Engineering design process; Software requirements specification; Stakeholder; TRIZ; Software requirements; Software design; Human–computer interaction; Software development; Programming language; Artificial intelligence; Psychology","score_opus":0.04554699575861654,"score_gpt":0.25256362273964506,"score_spread":0.20701662698102852,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3199338376","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99189585,0.00007619226,0.004127282,0.00020544461,0.0000072263624,0.000031806398,0.0000055074447,0.000025078618,0.0036255387],"genre_scores_gemma":[0.9965037,0.00004424615,0.0029414033,0.000054500873,0.0000020306832,0.000045251334,0.000009853161,0.000013740388,0.00038527587],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9743378,0.019997193,0.0009974856,0.00095234497,0.0032540392,0.0004611256],"domain_scores_gemma":[0.72762007,0.2457075,0.014340013,0.007153323,0.004034563,0.0011444296],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.014650838,0.0002803975,0.0002479427,0.00050267775,0.00089614233,0.003057997,0.0009969492,0.0009957255,0.0018272593],"category_scores_gemma":[0.13972901,0.00045671605,0.0004019563,0.000419308,0.0018695242,0.002632052,0.0020843013,0.0010473861,0.00020300069],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004229626,0.0027870205,0.28293785,0.0017381458,0.00038733895,0.0014963755,0.2691548,0.011133245,0.10999802,0.043786906,0.001836488,0.27051425],"study_design_scores_gemma":[0.0013565888,0.009773193,0.51662886,0.0013975347,0.0009353025,0.0034970923,0.13921969,0.07641877,0.13702187,0.083053455,0.030012105,0.00068548636],"about_ca_topic_score_codex":0.0011762147,"about_ca_topic_score_gemma":0.0012688629,"teacher_disagreement_score":0.9853492,"about_ca_system_score_codex":0.001043153,"about_ca_system_score_gemma":0.0012764023,"threshold_uncertainty_score":0.077481985},"labels":[],"label_agreement":null},{"id":"W3204491293","doi":"10.1109/tse.2021.3115772","title":"Characterizing and Mitigating Self-Admitted Technical Debt in Build Systems","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Technical debt; Software engineering; Operating system; Software; Software development","score_opus":0.009912218925696303,"score_gpt":0.22976416783745449,"score_spread":0.2198519489117582,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3204491293","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97298616,0.0004298349,0.02413975,0.0003070307,0.000017975479,0.00011710937,0.00028781014,0.00045680697,0.0012575954],"genre_scores_gemma":[0.9773691,0.00019494034,0.020488145,0.00009818897,0.000023231112,0.00008537016,0.0008665655,0.00013167498,0.00074275606],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.988746,0.0035318118,0.0013833623,0.0013539225,0.0043203407,0.0006645067],"domain_scores_gemma":[0.89648736,0.053107955,0.029121123,0.005825181,0.0139285335,0.0015298559],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007643808,0.0006584677,0.00048022694,0.006545451,0.001349102,0.0028056577,0.00081694825,0.0011136476,0.0004554142],"category_scores_gemma":[0.07248676,0.0005456097,0.0004571172,0.0029863508,0.0011942988,0.004127734,0.0027876757,0.0011208897,0.00028934507],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022871475,0.00019420395,0.7880252,0.0009881782,0.00009782507,0.0012724922,0.035808932,0.0041975486,0.023274442,0.0015321998,0.0019160134,0.1424642],"study_design_scores_gemma":[0.000022409766,0.00043131434,0.8448782,0.00065268343,0.00014737084,0.002309386,0.024977144,0.0873808,0.017601002,0.0048952335,0.0165105,0.0001939843],"about_ca_topic_score_codex":0.0052863266,"about_ca_topic_score_gemma":0.0102347005,"teacher_disagreement_score":0.007643808,"about_ca_system_score_codex":0.0013353741,"about_ca_system_score_gemma":0.001535728,"threshold_uncertainty_score":0.040424764},"labels":[],"label_agreement":null},{"id":"W3207487452","doi":"10.1109/tse.2021.3117966","title":"Pluto: Exposing Vulnerabilities in Inter-Contract Scenarios","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Reachability; Pluto; Symbolic execution; Computer security; Smart contract; Vulnerability (computing); False positive paradox; Fuzz testing; Programming language; Artificial intelligence; Theoretical computer science; Software","score_opus":0.008612286267831283,"score_gpt":0.21405317482825034,"score_spread":0.20544088856041906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3207487452","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38233876,0.003933374,0.33610395,0.0021647848,0.00016241694,0.0005131214,0.009510503,0.25457242,0.010700668],"genre_scores_gemma":[0.73686844,0.0011145936,0.23732968,0.0009204119,0.000043337466,0.00026171328,0.014236717,0.006231758,0.002993494],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99661726,0.00088190223,0.00023189725,0.00061317784,0.0012873043,0.00036853654],"domain_scores_gemma":[0.9930814,0.0036046256,0.0009971447,0.0016484468,0.0004966952,0.00017171344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025992848,0.0017033257,0.00047196035,0.003712466,0.0006596577,0.0014382647,0.0017867933,0.0016203168,0.0019147041],"category_scores_gemma":[0.013580611,0.00082011137,0.001322094,0.0015339785,0.0013606034,0.0052404744,0.0031784652,0.0014249546,0.000663128],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013323867,0.0006459742,0.15160431,0.0026434017,0.0006250866,0.003960615,0.0031799907,0.1462241,0.055091668,0.036755845,0.07803425,0.51990235],"study_design_scores_gemma":[0.00022981987,0.000424122,0.02979387,0.00048949564,0.0002619929,0.0026123931,0.0007393955,0.802127,0.059268095,0.039396744,0.06442049,0.00023668206],"about_ca_topic_score_codex":0.008151047,"about_ca_topic_score_gemma":0.0102044055,"teacher_disagreement_score":0.008151047,"about_ca_system_score_codex":0.0011673683,"about_ca_system_score_gemma":0.0019523305,"threshold_uncertainty_score":0.016207218},"labels":[],"label_agreement":null},{"id":"W3210529140","doi":"10.1109/tse.2022.3171295","title":"Fragment-Based Test Generation for Web Apps","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Test (biology); Fragment (logic); Web application; Software engineering; World Wide Web; Programming language; Operating system; Database","score_opus":0.02069671311464428,"score_gpt":0.22918178894011498,"score_spread":0.2084850758254707,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3210529140","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09182589,0.00019487993,0.85309553,0.00033366177,0.00006304708,0.0005043897,0.0011522243,0.049281936,0.0035484405],"genre_scores_gemma":[0.6738239,0.00013226755,0.31693915,0.00030162334,0.000029189663,0.0005612797,0.0033628964,0.0026226728,0.0022271343],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99674964,0.0010663024,0.00021343853,0.0004718731,0.0012493642,0.00024939125],"domain_scores_gemma":[0.9904704,0.0055458914,0.00057092553,0.0021016144,0.0011485941,0.00016261076],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016384395,0.0012098743,0.00056859513,0.0014871733,0.0003385599,0.0008740958,0.0023680767,0.0010646315,0.0047206716],"category_scores_gemma":[0.014549616,0.0005574336,0.0011243242,0.0007223791,0.0008969534,0.0018954085,0.0014657206,0.0011239557,0.00092873466],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013874973,0.0006958538,0.0116729075,0.0005454867,0.00016974116,0.0012446387,0.00051960786,0.4353633,0.06595454,0.021728596,0.019302122,0.4414158],"study_design_scores_gemma":[0.00009465532,0.0002910358,0.0008719799,0.0000310649,0.000033482127,0.00022962329,0.00003199759,0.9532752,0.031603504,0.010059671,0.0034465198,0.000031245134],"about_ca_topic_score_codex":0.0042573754,"about_ca_topic_score_gemma":0.003842487,"teacher_disagreement_score":0.0047206716,"about_ca_system_score_codex":0.0010411941,"about_ca_system_score_gemma":0.001331866,"threshold_uncertainty_score":0.015792191},"labels":[],"label_agreement":null},{"id":"W4212906466","doi":"10.1109/tse.2022.3152148","title":"An Empirical Study of Yanked Releases in the Rust Package Registry","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Alberta","funders":"University of Alberta","keywords":"Computer science; Rust (programming language); Software versioning; Dependency (UML); Software package; Code (set theory); Software release life cycle; Software; Software engineering; Operating system; World Wide Web; Database; Programming language; Software development; Software quality","score_opus":0.024242124774664125,"score_gpt":0.28821594235437353,"score_spread":0.2639738175797094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4212906466","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99674857,0.00017423982,0.0004091445,0.00024252894,0.0000072657062,0.00002270318,0.0002075879,0.000016116077,0.0021719784],"genre_scores_gemma":[0.9979538,0.00022268597,0.00045258683,0.00010124382,0.000010338625,0.000028965374,0.00041694293,0.000031171985,0.0007823065],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.991971,0.0032152534,0.00079500943,0.0010860183,0.002113384,0.0008192302],"domain_scores_gemma":[0.8389564,0.08298042,0.053733166,0.006711903,0.012932061,0.0046860725],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010675109,0.00031199236,0.00032187812,0.0023763336,0.0012435244,0.0029841196,0.0011374695,0.0007748754,0.0034705915],"category_scores_gemma":[0.059205726,0.00045752173,0.00043391535,0.003779939,0.0022300347,0.007493709,0.0021044472,0.0026177072,0.0009904249],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015426023,0.00029116252,0.97039056,0.00012972484,0.000041971787,0.0006132322,0.012357914,0.00029988948,0.00040505722,0.0015234423,0.0014576991,0.012335141],"study_design_scores_gemma":[0.000017940656,0.0002278944,0.95045483,0.00015332489,0.000038083665,0.00067633274,0.03697499,0.0023937204,0.00052427134,0.0005498221,0.007936163,0.000052524992],"about_ca_topic_score_codex":0.0081381425,"about_ca_topic_score_gemma":0.009612053,"teacher_disagreement_score":0.010675109,"about_ca_system_score_codex":0.0016107208,"about_ca_system_score_gemma":0.0014575794,"threshold_uncertainty_score":0.05645603},"labels":[],"label_agreement":null},{"id":"W4213075374","doi":"10.1109/tse.2018.2872711","title":"An Interactive and Dynamic Search-Based Approach to Software Refactoring Recommendations","year":2020,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":64,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Ford Motor Company","keywords":"Code refactoring; Computer science; Software; Software engineering; Software evolution; Software quality; Software system; Set (abstract data type); Process (computing); Merge (version control); Software metric; Software development; Programming language; Software construction; Information retrieval","score_opus":0.0253112702250691,"score_gpt":0.27451925036300423,"score_spread":0.24920798013793513,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4213075374","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032135043,0.0010077901,0.9515459,0.0007706113,0.000100619305,0.00080377574,0.00036259668,0.00823389,0.00503971],"genre_scores_gemma":[0.2172736,0.00040615475,0.7751564,0.0005705136,0.00010554547,0.00089181354,0.0010267282,0.00037450928,0.0041948194],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964857,0.0012383852,0.00024322786,0.00073873595,0.0010675116,0.00022634993],"domain_scores_gemma":[0.9916998,0.005499951,0.0005089464,0.0006358892,0.0013524641,0.000303015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037344655,0.0026258468,0.002270312,0.004845952,0.0012129529,0.0016772674,0.0061908565,0.0038290918,0.005104146],"category_scores_gemma":[0.0116998,0.0012718793,0.0016227376,0.0027513104,0.00079587137,0.002129598,0.0021969227,0.0019883174,0.001728254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081898854,0.0019836363,0.0073219878,0.00092755875,0.0006001513,0.000634213,0.0013508055,0.30628523,0.0225995,0.0072470196,0.017061409,0.6331694],"study_design_scores_gemma":[0.00015204285,0.00024034463,0.0008140076,0.000046521673,0.00013754409,0.00015655298,0.00013725007,0.9861704,0.002659017,0.004143494,0.005287973,0.00005479941],"about_ca_topic_score_codex":0.011082807,"about_ca_topic_score_gemma":0.028010545,"teacher_disagreement_score":0.011082807,"about_ca_system_score_codex":0.0011717823,"about_ca_system_score_gemma":0.0020108798,"threshold_uncertainty_score":0.022036552},"labels":[],"label_agreement":null},{"id":"W4214643715","doi":"10.1109/tse.2022.3154672","title":"An Empirical Study on Log Level Prediction for Multi-Component Systems","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Component (thermodynamics); Interpretability; Computer science; Logging; Component-based software engineering; Data mining; Leverage (statistics); Software; Software system; Machine learning; Operating system","score_opus":0.05913968027635651,"score_gpt":0.29785239876476993,"score_spread":0.2387127184884134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214643715","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9704028,0.0017694631,0.022994746,0.0009568747,0.00007049367,0.0001305899,0.0013722442,0.0005239247,0.0017789348],"genre_scores_gemma":[0.9882625,0.00030350298,0.00787029,0.0000960709,0.000041973435,0.00006479872,0.0028159148,0.000073422365,0.00047155897],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98868644,0.0058370503,0.00085741293,0.0022026852,0.0019545183,0.00046192767],"domain_scores_gemma":[0.7747768,0.1928617,0.008856159,0.011234486,0.010323417,0.0019473545],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021519389,0.0011821694,0.0008320941,0.0021376142,0.0007669495,0.0024740596,0.0017382387,0.0013373449,0.0011832967],"category_scores_gemma":[0.11350352,0.0004421207,0.0010174109,0.002862363,0.0011697024,0.005189698,0.0011711513,0.0041371887,0.0005781832],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00087737635,0.001475309,0.7648036,0.000720404,0.0005290651,0.0004278326,0.0016477695,0.12289462,0.00092844025,0.0019868894,0.008036801,0.09567189],"study_design_scores_gemma":[0.00006548003,0.00071197585,0.18244477,0.0001661172,0.0001458844,0.00036295134,0.0012609072,0.8052325,0.0016566158,0.0032826609,0.004602946,0.000067105386],"about_ca_topic_score_codex":0.0126874475,"about_ca_topic_score_gemma":0.008486643,"teacher_disagreement_score":0.021519389,"about_ca_system_score_codex":0.0016735308,"about_ca_system_score_gemma":0.0010253872,"threshold_uncertainty_score":0.113806784},"labels":[],"label_agreement":null},{"id":"W4220988444","doi":"10.1109/tse.2022.3162236","title":"Selecting Context-Sensitivity Modularly for Accelerating Object-Sensitive Pointer Analysis","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Notation; Pointer (user interface); Pointer analysis; Computer science; Programming language; Context (archaeology); Theoretical computer science; Algorithm; Mathematics; Static analysis; Artificial intelligence; Arithmetic","score_opus":0.019674976284096056,"score_gpt":0.23646358962900285,"score_spread":0.2167886133449068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220988444","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03563252,0.00060597627,0.93242145,0.0003419876,0.00014041971,0.00019838323,0.00011650988,0.023167111,0.0073756864],"genre_scores_gemma":[0.28831822,0.00058103196,0.6974962,0.0007766374,0.00013113495,0.00031186204,0.00042807017,0.005351834,0.006605095],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99883705,0.00019824877,0.00009048816,0.0002804849,0.00040153248,0.00019226791],"domain_scores_gemma":[0.99726105,0.0010589613,0.00022663883,0.0009534893,0.00038943544,0.00011038555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011559201,0.0014759016,0.0007449268,0.0011011709,0.0006196414,0.0020635563,0.0022517487,0.00076743675,0.006261737],"category_scores_gemma":[0.0059754737,0.00072352146,0.0010134258,0.0009194915,0.001339947,0.00336849,0.003178358,0.0021012563,0.0029480893],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083806604,0.00032949945,0.007828146,0.00083960884,0.00016276202,0.0008303908,0.0016421538,0.030381147,0.23681204,0.11563222,0.018272497,0.58643144],"study_design_scores_gemma":[0.00016929732,0.00033799213,0.0027561777,0.00034161966,0.00025958833,0.0007360967,0.0003533719,0.41738036,0.398335,0.089403756,0.08969143,0.0002353336],"about_ca_topic_score_codex":0.0013644047,"about_ca_topic_score_gemma":0.0025797444,"teacher_disagreement_score":0.006261737,"about_ca_system_score_codex":0.0008265781,"about_ca_system_score_gemma":0.0021797162,"threshold_uncertainty_score":0.020947576},"labels":[],"label_agreement":null},{"id":"W4225878264","doi":"10.1109/tse.2022.3162985","title":"Static Profiling of Alloy Models","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Correctness; Modeling language; Natural language processing; Profiling (computer programming); Matching (statistics); Programming language; Software; Artificial intelligence; Data science; Software engineering","score_opus":0.02111579023518419,"score_gpt":0.23700404072038123,"score_spread":0.21588825048519703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225878264","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4088794,0.0012068754,0.5074705,0.00089024764,0.0001885393,0.00047155778,0.008408655,0.022092577,0.05039166],"genre_scores_gemma":[0.7159907,0.0007045236,0.24422202,0.00023933478,0.000057883895,0.00042207062,0.01848922,0.005217573,0.014656664],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9956151,0.0011145442,0.0003516233,0.0005931279,0.0020762787,0.00024923947],"domain_scores_gemma":[0.99038386,0.0042041535,0.000620357,0.0023121359,0.002358592,0.00012100979],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002538957,0.0007842832,0.0007539708,0.0034086152,0.0011508228,0.003000672,0.0013590934,0.00088660617,0.0057470277],"category_scores_gemma":[0.018309066,0.0008848462,0.0012605661,0.002954527,0.0007066035,0.0036012856,0.0015728478,0.0008747418,0.002338997],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010946877,0.00043049792,0.06636906,0.0013870876,0.0002715959,0.0016561463,0.011680538,0.200815,0.056666806,0.21577603,0.031670127,0.41218242],"study_design_scores_gemma":[0.000041414754,0.00017260328,0.010080153,0.00017276945,0.00014973554,0.00060785626,0.0014854175,0.7923206,0.049567506,0.035665393,0.109631285,0.000105273575],"about_ca_topic_score_codex":0.008989217,"about_ca_topic_score_gemma":0.014743637,"teacher_disagreement_score":0.008989217,"about_ca_system_score_codex":0.0021015161,"about_ca_system_score_gemma":0.0019701284,"threshold_uncertainty_score":0.019225717},"labels":[],"label_agreement":null},{"id":"W4226137778","doi":"10.1109/tse.2022.3201209","title":"Flakify: A Black-Box, Language Model-Based Predictor for Flaky Tests","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; Canada Research Chairs; Western Canada Research Grid; Compute Canada","keywords":"Computer science; Debugging; Code coverage; Code (set theory); Set (abstract data type); Source code; White-box testing; Programming language; Software; Overhead (engineering); Black box; Test (biology); Machine learning; Artificial intelligence; Software development; Software construction","score_opus":0.013714965914711835,"score_gpt":0.24877966200787097,"score_spread":0.23506469609315914,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226137778","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.089647606,0.0015246393,0.8467523,0.0011428255,0.00027746885,0.00042032055,0.003173421,0.054260246,0.0028011806],"genre_scores_gemma":[0.645858,0.0005903872,0.33384275,0.0009961568,0.00020989793,0.00087209354,0.008829081,0.0020843896,0.006717304],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99712914,0.0005796375,0.00018284681,0.0006614553,0.0010890242,0.00035796425],"domain_scores_gemma":[0.98890173,0.007022789,0.0015559274,0.0006268752,0.0013899634,0.00050266995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028134543,0.0030255343,0.0015329396,0.003036439,0.00049985107,0.0017852555,0.0019783408,0.0015846586,0.004328669],"category_scores_gemma":[0.017843857,0.00071288645,0.0013143352,0.00097393105,0.00093848124,0.002848697,0.0024710651,0.0038291665,0.0031327147],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018850644,0.001275659,0.07272731,0.00072494894,0.00033295137,0.0007492206,0.0003475074,0.3553134,0.025055079,0.0058262427,0.023767522,0.5119951],"study_design_scores_gemma":[0.00004852785,0.0002759219,0.0026703598,0.000057570243,0.000045198045,0.00009786372,0.000030690248,0.98430485,0.007010586,0.0030858025,0.0023254715,0.00004718559],"about_ca_topic_score_codex":0.007705452,"about_ca_topic_score_gemma":0.010190872,"teacher_disagreement_score":0.007705452,"about_ca_system_score_codex":0.0011848464,"about_ca_system_score_gemma":0.0035988155,"threshold_uncertainty_score":0.015321195},"labels":[],"label_agreement":null},{"id":"W4247801608","doi":"10.1109/tse.2021.3109563","title":"Analyzing Android Taint Analysis Tools: FlowDroid, Amandroid, and DroidSafe","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Rendering (computer graphics); Android (operating system); Set (abstract data type); Data mining; Information retrieval; Artificial intelligence; Programming language; Operating system","score_opus":0.008880119046626464,"score_gpt":0.2199095884629682,"score_spread":0.21102946941634174,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4247801608","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7230197,0.009707817,0.104343995,0.001007694,0.00077305076,0.001185484,0.007969615,0.13490921,0.017083388],"genre_scores_gemma":[0.8756294,0.0016626844,0.09940128,0.0004929712,0.00012335663,0.000630111,0.011943429,0.0044765845,0.0056400863],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9953603,0.00056634436,0.00045027598,0.000743936,0.0024377955,0.00044138142],"domain_scores_gemma":[0.98995215,0.0044240346,0.0010665252,0.0019529491,0.0023198912,0.00028455377],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021580062,0.002619528,0.00084551086,0.00559045,0.0007646737,0.0015726914,0.0019189834,0.0009764875,0.0014557124],"category_scores_gemma":[0.012926981,0.00066231575,0.0011171071,0.0025288814,0.00088107376,0.0032123055,0.0017954321,0.0013834244,0.0006842689],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025144161,0.0012901193,0.08296035,0.0042784056,0.00082620565,0.0011719803,0.0022440504,0.03445578,0.050957505,0.008140735,0.055562243,0.75559825],"study_design_scores_gemma":[0.0007625829,0.0030793876,0.14316322,0.0010147995,0.0008575327,0.0031474775,0.0021117588,0.5467716,0.18775366,0.008105048,0.10225231,0.0009806701],"about_ca_topic_score_codex":0.009781445,"about_ca_topic_score_gemma":0.0119747035,"teacher_disagreement_score":0.009781445,"about_ca_system_score_codex":0.00094419747,"about_ca_system_score_gemma":0.0018582447,"threshold_uncertainty_score":0.019449055},"labels":[],"label_agreement":null},{"id":"W4285121201","doi":"10.1109/tse.2022.3188005","title":"Automated Generation and Evaluation of JMH Microbenchmark Suites From Unit Tests","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Unit testing; Programming language; Eclipse; Software; Operating system","score_opus":0.023174269775778533,"score_gpt":0.2505865902937494,"score_spread":0.2274123205179709,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285121201","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8166045,0.0011186003,0.086387224,0.00047805675,0.00032718747,0.00078295165,0.00415437,0.08231437,0.007832728],"genre_scores_gemma":[0.79147404,0.00034137993,0.17963354,0.00038717993,0.00008556143,0.00067114946,0.018111875,0.006828784,0.0024665368],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99296016,0.0019070759,0.0005385617,0.0011213602,0.0030056941,0.00046711334],"domain_scores_gemma":[0.9745418,0.012079475,0.00229532,0.0060484125,0.0043895002,0.0006453917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057897847,0.0014490866,0.00070906844,0.0026163955,0.0004994577,0.0011721172,0.002948661,0.00086977886,0.0016402807],"category_scores_gemma":[0.0312687,0.00060381007,0.0007512953,0.001334767,0.001041648,0.0013149625,0.0014437044,0.0013238809,0.00094456243],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001828546,0.003377959,0.06812569,0.0026545378,0.00052980037,0.0016263922,0.0018407148,0.18888275,0.14063391,0.00790738,0.05561858,0.5269738],"study_design_scores_gemma":[0.00061840325,0.0017663074,0.048832014,0.00017475068,0.0001244542,0.0006469121,0.00048669835,0.66413206,0.25801823,0.003762542,0.021233281,0.00020438684],"about_ca_topic_score_codex":0.0029484606,"about_ca_topic_score_gemma":0.0038482808,"teacher_disagreement_score":0.0057897847,"about_ca_system_score_codex":0.0010970149,"about_ca_system_score_gemma":0.0015541273,"threshold_uncertainty_score":0.03061968},"labels":[],"label_agreement":null},{"id":"W4285169595","doi":"10.1109/tse.2022.3175752","title":"Annotative Software Product Line Analysis Using Variability-Aware Datalog","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"General Motors of Canada; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Datalog; Computer science; Programming language; Software product line; Java; Software; Product (mathematics); Pointer (user interface); Inference; Software engineering; Theoretical computer science; Software development; Artificial intelligence","score_opus":0.03845076192181567,"score_gpt":0.28467542222655134,"score_spread":0.24622466030473567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285169595","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02464332,0.000114774564,0.96192145,0.00015497101,0.00002357249,0.000055513843,0.00044853767,0.011671305,0.00096651074],"genre_scores_gemma":[0.3104195,0.00025893134,0.68255806,0.00022069766,0.0000485303,0.000117672535,0.0021617631,0.0026522977,0.0015626126],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99675035,0.00065585936,0.00021714644,0.0006255876,0.00159322,0.00015792105],"domain_scores_gemma":[0.9832653,0.009418263,0.0011611611,0.0042140223,0.0018069488,0.0001342884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025405067,0.0009752988,0.00066255126,0.0018830282,0.00068090565,0.0022641444,0.0021637483,0.00082903175,0.0022943965],"category_scores_gemma":[0.014334268,0.00079451915,0.0016146685,0.0012638223,0.001329543,0.0038871197,0.002334305,0.0020755422,0.000759045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004723296,0.00034486508,0.020771958,0.0011019066,0.00024796117,0.0012871027,0.0019886503,0.22759712,0.07247082,0.07078676,0.010147383,0.59278315],"study_design_scores_gemma":[0.00002962262,0.00008628459,0.0018477556,0.000115292976,0.00006989619,0.00030513885,0.00023632568,0.8345209,0.059491098,0.08816421,0.015074975,0.000058502017],"about_ca_topic_score_codex":0.0030085107,"about_ca_topic_score_gemma":0.0052390005,"teacher_disagreement_score":0.0030085107,"about_ca_system_score_codex":0.0008938531,"about_ca_system_score_gemma":0.0016913869,"threshold_uncertainty_score":0.013435662},"labels":[],"label_agreement":null},{"id":"W4285194931","doi":"10.1109/tse.2022.3177228","title":"SCS-Gan: Learning Functionality-Agnostic Stylometric Representations for Source Code Authorship Verification","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Source code; Code (set theory); Set (abstract data type); Task (project management); Benchmark (surveying); Malware; Artificial intelligence; Adversarial system; Representation (politics); Information retrieval; Machine learning; Natural language processing; Programming language; Computer security","score_opus":0.03621979625623392,"score_gpt":0.2699609822255617,"score_spread":0.23374118596932777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285194931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08863712,0.0013582802,0.8964856,0.0010159237,0.0002756255,0.00018341767,0.0008063173,0.0051563624,0.0060813],"genre_scores_gemma":[0.8904487,0.0004877955,0.09579516,0.0008220358,0.00017320979,0.00022468195,0.002022126,0.00029727852,0.009728984],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993262,0.000244231,0.00002856495,0.00017288061,0.00014901976,0.00007903993],"domain_scores_gemma":[0.9977843,0.0013205621,0.00019236607,0.00030197296,0.00031290957,0.00008790781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001365232,0.0010432021,0.00083879346,0.0012037198,0.0002726062,0.0006426117,0.0014338101,0.0011914734,0.0020826873],"category_scores_gemma":[0.005501262,0.0004116346,0.000807097,0.0006675985,0.0009346869,0.0011559047,0.00096324174,0.0016738671,0.000972315],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035230495,0.0002048664,0.0068629635,0.00019029656,0.00014370614,0.0002699919,0.00020628147,0.63605237,0.008430443,0.0126525955,0.012358544,0.32227564],"study_design_scores_gemma":[0.000005720016,0.000015761329,0.0002603795,0.00000799721,0.0000062811064,0.000029592458,0.000005666822,0.9945485,0.0008261348,0.0038181022,0.00047056752,0.0000053158105],"about_ca_topic_score_codex":0.0026557914,"about_ca_topic_score_gemma":0.0035162352,"teacher_disagreement_score":0.0026557914,"about_ca_system_score_codex":0.0009865382,"about_ca_system_score_gemma":0.0007618335,"threshold_uncertainty_score":0.0072200894},"labels":[],"label_agreement":null},{"id":"W4288391571","doi":"10.1109/tse.2022.3194640","title":"FalsifAI: Falsification of AI-Enabled Hybrid Control Systems Guided by Time-Aware Coverage Criteria","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Correctness; Robustness (evolution); Artificial neural network; Notation; Context (archaeology); Hybrid system; Semantics (computer science); Artificial intelligence; Model checking; Theoretical computer science; Programming language; Machine learning","score_opus":0.007528728385960206,"score_gpt":0.225826501456591,"score_spread":0.2182977730706308,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288391571","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051563036,0.00029336946,0.9412859,0.0004476713,0.00007161962,0.000116484276,0.00018334536,0.0014353058,0.004603367],"genre_scores_gemma":[0.90059584,0.00016731686,0.096763074,0.0002096693,0.000047060814,0.00018378143,0.00027669646,0.00023270534,0.0015238712],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99705184,0.00075209816,0.00018931383,0.00055337616,0.0010151777,0.00043816253],"domain_scores_gemma":[0.9888131,0.008006147,0.0009461608,0.0008778189,0.0010519265,0.00030480538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031500298,0.0013547093,0.0009052843,0.0012095106,0.0006931254,0.0021001077,0.0015694025,0.0014079623,0.002868589],"category_scores_gemma":[0.01687434,0.0004162077,0.0015078865,0.00041263996,0.0032570176,0.0019553467,0.0026180795,0.00153952,0.00027263063],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042096118,0.000086822794,0.0048733703,0.00033675547,0.0001483546,0.0009361655,0.0005429609,0.78415513,0.0155335525,0.15188627,0.0015379753,0.039541688],"study_design_scores_gemma":[0.000019978017,0.00007529673,0.00024370934,0.000042201886,0.000019559926,0.000092935166,0.000045165092,0.94757956,0.005262566,0.045688737,0.00091135164,0.000018856488],"about_ca_topic_score_codex":0.0036730082,"about_ca_topic_score_gemma":0.0022805203,"teacher_disagreement_score":0.0036730082,"about_ca_system_score_codex":0.0015598461,"about_ca_system_score_gemma":0.0015561514,"threshold_uncertainty_score":0.01665914},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"not_applicable","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"software","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W4296338325","doi":"10.1109/tse.2022.3207428","title":"Code Cloning in Smart Contracts on the Ethereum Platform: An Extended Replication Study","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Université de Montréal; McGill University","funders":"","keywords":"Solidity; Computer science; Cloning (programming); clone (Java method); Context (archaeology); Code (set theory); Granularity; Source code; Software engineering; Programming language; Code reuse; Source lines of code; Replication (statistics); Software; Mathematics","score_opus":0.026244746252368042,"score_gpt":0.26745966406046645,"score_spread":0.2412149178080984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4296338325","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9823293,0.00045600955,0.015092302,0.00010883632,0.000011682652,0.00011504959,0.0002523854,0.00025012007,0.0013842826],"genre_scores_gemma":[0.97762567,0.0002750568,0.019282231,0.000072015355,0.00001785232,0.00012053173,0.00086430344,0.00018325688,0.0015590776],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98970175,0.0033752958,0.00075727655,0.0020832932,0.0034185108,0.00066389143],"domain_scores_gemma":[0.8980442,0.052077215,0.012790357,0.02433307,0.011767262,0.0009878876],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.010121847,0.00047217542,0.00063347426,0.0031230873,0.0011181894,0.00154257,0.0016732567,0.0010324519,0.0012329955],"category_scores_gemma":[0.060795043,0.00043597393,0.0008741361,0.0030674634,0.0018576541,0.0036046014,0.0021165058,0.0011929071,0.00045506688],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008554739,0.0009932367,0.7043202,0.00075580174,0.0004810419,0.0035308634,0.01230916,0.046979733,0.026565999,0.017151367,0.0029828793,0.18307424],"study_design_scores_gemma":[0.00020257843,0.002183728,0.39304477,0.00047583625,0.00061491376,0.008041547,0.012898946,0.47202238,0.04383896,0.028769502,0.03752086,0.00038599552],"about_ca_topic_score_codex":0.0062007923,"about_ca_topic_score_gemma":0.004208506,"teacher_disagreement_score":0.9898782,"about_ca_system_score_codex":0.0012867949,"about_ca_system_score_gemma":0.00165405,"threshold_uncertainty_score":0.053530097},"labels":[],"label_agreement":null},{"id":"W4312260323","doi":"10.1109/tse.2022.3217544","title":"Dynamic Human-in-the-Loop Assertion Generation","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of British Columbia","funders":"","keywords":"Assertion; Computer science; Programming language; TypeScript; JavaScript; Test (biology); Test case; Workflow; Notation; Automation; Software engineering; Variable (mathematics); Database; Arithmetic; Mathematics; Machine learning","score_opus":0.02170916402304557,"score_gpt":0.2527178957400945,"score_spread":0.23100873171704894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312260323","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08029779,0.00014883779,0.82911384,0.00025825226,0.00019483402,0.00090489525,0.0017700059,0.08249565,0.004815862],"genre_scores_gemma":[0.36279193,0.00013704173,0.61447996,0.00035420002,0.00006843639,0.0011454072,0.0052591586,0.010457136,0.0053067775],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99501747,0.0014787145,0.00040174296,0.001361619,0.001425934,0.00031457323],"domain_scores_gemma":[0.96159846,0.026729865,0.0019503839,0.0050847474,0.0041812467,0.00045528225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0063063125,0.0013946992,0.00054005353,0.0016744942,0.0003932909,0.0013510083,0.0019337161,0.0007406588,0.006553347],"category_scores_gemma":[0.042927235,0.00075302913,0.0008070478,0.0005642779,0.0009706139,0.0015971626,0.0017751135,0.0012127737,0.0029760385],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012664472,0.0011484748,0.04237473,0.0016338762,0.00022429395,0.003942758,0.0057012117,0.06717591,0.08944133,0.029774154,0.05652895,0.7007878],"study_design_scores_gemma":[0.00035865276,0.0005776924,0.0065827863,0.0003688415,0.00013320209,0.0018594358,0.00060668547,0.6834618,0.19675201,0.026169928,0.08294416,0.00018474436],"about_ca_topic_score_codex":0.0011430167,"about_ca_topic_score_gemma":0.0013313796,"teacher_disagreement_score":0.006553347,"about_ca_system_score_codex":0.00048670958,"about_ca_system_score_gemma":0.0014916699,"threshold_uncertainty_score":0.033351302},"labels":[],"label_agreement":null},{"id":"W4312614778","doi":"10.1109/tse.2022.3222160","title":"Studying the Interplay Between the Durations and Breakages of Continuous Integration Builds","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Toronto; University of Ottawa","funders":"","keywords":"Computer science; Context (archaeology); World Wide Web; Data science; Information retrieval","score_opus":0.013719677836531744,"score_gpt":0.2563637641058254,"score_spread":0.24264408626929365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312614778","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9923495,0.00026572376,0.003710815,0.00016884842,0.000010067904,0.00005730786,0.0001362778,0.000042678617,0.0032587692],"genre_scores_gemma":[0.9955857,0.0001950517,0.0032700482,0.000043262124,0.000010835827,0.00012041439,0.00015789637,0.000022928554,0.0005938914],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9840295,0.0071409363,0.0015811999,0.0022142772,0.0040776734,0.00095646875],"domain_scores_gemma":[0.6942674,0.20598488,0.070065804,0.009284007,0.014432425,0.005965609],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014369593,0.00046721558,0.00034320707,0.0024662707,0.00091743446,0.002953742,0.00115902,0.00094532093,0.002723546],"category_scores_gemma":[0.13073301,0.00076737313,0.00035262044,0.002174781,0.0013273554,0.003908128,0.0021171707,0.0015994825,0.00039513365],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007166176,0.0015479302,0.8614068,0.00067042763,0.00029594297,0.0002444226,0.037270047,0.0021682696,0.008247212,0.003139944,0.0008379386,0.083454415],"study_design_scores_gemma":[0.000039485774,0.0012268093,0.9672998,0.00017463794,0.00015235372,0.00016307241,0.01812942,0.0027430514,0.0033984033,0.001890599,0.0046955566,0.00008677212],"about_ca_topic_score_codex":0.003868722,"about_ca_topic_score_gemma":0.00835566,"teacher_disagreement_score":0.014369593,"about_ca_system_score_codex":0.0017420199,"about_ca_system_score_gemma":0.0017793187,"threshold_uncertainty_score":0.07599455},"labels":[],"label_agreement":null},{"id":"W4312834255","doi":"10.1109/tse.2022.3213041","title":"Data-Driven Mutation Analysis for Cyber-Physical Systems","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission; European Space Agency","keywords":"Computer science; Interoperability; Test suite; Set (abstract data type); Data mining; Suite; Mutation; Software; Quality (philosophy); Programming language; Software engineering; Theoretical computer science; Test case; Machine learning; World Wide Web","score_opus":0.03462965960977726,"score_gpt":0.27028679770145464,"score_spread":0.23565713809167738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312834255","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055836827,0.00019343795,0.93877953,0.0002169963,0.000043329695,0.00010649741,0.00026833007,0.0031812924,0.001373741],"genre_scores_gemma":[0.6022233,0.00023985197,0.39436474,0.00017519435,0.00003158809,0.00029629093,0.00086492335,0.00039899402,0.001405177],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997514,0.000590904,0.00014869292,0.0003800354,0.0012150757,0.00015141409],"domain_scores_gemma":[0.9933924,0.004318061,0.0006096191,0.000463956,0.0010688425,0.00014704758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018785819,0.001152268,0.00076801993,0.0037234523,0.0004882949,0.0012552721,0.001282984,0.000888247,0.0012705765],"category_scores_gemma":[0.009719999,0.00033946632,0.00168139,0.0009881011,0.0014551114,0.0012150812,0.0011102479,0.001311142,0.00025580855],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021533576,0.00024923676,0.008937936,0.00031762375,0.00016243673,0.00092630525,0.00023957509,0.733138,0.04398169,0.061809678,0.0017691959,0.14825292],"study_design_scores_gemma":[0.000016686372,0.000044299424,0.0005051692,0.000013784274,0.000015453505,0.00009733525,0.000016505406,0.9732617,0.012465496,0.012771644,0.0007725784,0.000019455392],"about_ca_topic_score_codex":0.0041515785,"about_ca_topic_score_gemma":0.0024889081,"teacher_disagreement_score":0.0041515785,"about_ca_system_score_codex":0.0014708437,"about_ca_system_score_gemma":0.0016988365,"threshold_uncertainty_score":0.010671794},"labels":[],"label_agreement":null},{"id":"W4313142332","doi":"10.1109/tse.2022.3220740","title":"A Comprehensive Investigation of the Impact of Class Overlap on Software Defect Prediction","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Science Foundation of Jiangsu Province; National Natural Science Foundation of China","keywords":"Computer science; Class (philosophy); Software; Data mining; Rank (graph theory); Feature (linguistics); Machine learning; Identification (biology); Artificial intelligence; Mathematics","score_opus":0.018022052625698195,"score_gpt":0.24269995323414043,"score_spread":0.22467790060844223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313142332","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9494919,0.0014135552,0.044702947,0.0003754034,0.000060106377,0.0000690631,0.0011643672,0.0010114581,0.0017112552],"genre_scores_gemma":[0.9770493,0.00025336634,0.019386556,0.00007075118,0.000030875788,0.00004235473,0.0027446705,0.00006720523,0.0003549891],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99612576,0.0008478395,0.00028111125,0.0011709228,0.0012792719,0.00029519826],"domain_scores_gemma":[0.9789549,0.013159288,0.0021362759,0.0028644467,0.0023116036,0.00057336927],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053457017,0.0010747955,0.0008930256,0.0026322287,0.00071458315,0.0013672692,0.0011623633,0.00089289626,0.00051210314],"category_scores_gemma":[0.020804431,0.00029112792,0.001038219,0.002044341,0.00068627344,0.0026446395,0.0013548855,0.0012556596,0.00026221533],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063045934,0.00066101947,0.4232097,0.00042778772,0.00055296835,0.0007890104,0.00064355525,0.25902307,0.008562104,0.0017360365,0.007582851,0.29618144],"study_design_scores_gemma":[0.000021394577,0.00040812956,0.102615416,0.00007691019,0.00010890874,0.0005580258,0.00050228235,0.88276607,0.007388038,0.0026023833,0.002906167,0.000046248177],"about_ca_topic_score_codex":0.006285302,"about_ca_topic_score_gemma":0.006757194,"teacher_disagreement_score":0.006285302,"about_ca_system_score_codex":0.00073660014,"about_ca_system_score_gemma":0.0009705519,"threshold_uncertainty_score":0.028271139},"labels":[],"label_agreement":null},{"id":"W4319663674","doi":"10.1109/tse.2023.3243522","title":"Black-Box Testing of Deep Neural Networks through Test Case Diversity","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":86,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Black box; Artificial neural network; White-box testing; Test (biology); Diversity (politics); Software testing; Artificial intelligence; Machine learning; Software engineering; Software; Programming language; Software development; Software construction","score_opus":0.02156941011280877,"score_gpt":0.23671552213532804,"score_spread":0.21514611202251926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319663674","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87533295,0.0011203757,0.11948566,0.0003360079,0.000051207804,0.000094778356,0.00052418467,0.0015249815,0.0015298729],"genre_scores_gemma":[0.97160983,0.000094975425,0.027123831,0.00008977945,0.000020749952,0.000084370324,0.0006373148,0.00009197805,0.0002472018],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9929224,0.0026434667,0.0006068602,0.001445215,0.001918922,0.00046309963],"domain_scores_gemma":[0.92885566,0.054989837,0.0063725286,0.0050754384,0.0035818666,0.0011246372],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065192985,0.0014387339,0.00084518024,0.0027192067,0.00038831812,0.0012053519,0.002055702,0.0012377034,0.0007586982],"category_scores_gemma":[0.046381094,0.0004723933,0.00079732656,0.0010948532,0.0013593095,0.0031075014,0.0018733774,0.0012428754,0.00016813156],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012906842,0.0005188225,0.10954087,0.00038698706,0.00051128474,0.00065588555,0.0005118351,0.6672203,0.015461934,0.005416331,0.002007034,0.196478],"study_design_scores_gemma":[0.000035263358,0.00029020253,0.0056914836,0.000035766654,0.00004316562,0.00015222379,0.00007504083,0.9787965,0.009196274,0.005246384,0.00041632436,0.00002144429],"about_ca_topic_score_codex":0.0029680692,"about_ca_topic_score_gemma":0.0041303043,"teacher_disagreement_score":0.0065192985,"about_ca_system_score_codex":0.001389049,"about_ca_system_score_gemma":0.0010316157,"threshold_uncertainty_score":0.03447777},"labels":[],"label_agreement":null},{"id":"W4320005455","doi":"10.1109/tse.2023.3242588","title":"Trace Diagnostics for Signal-Based Temporal Properties","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"TRACE (psycholinguistics); Computer science; Digital subscriber line; Property (philosophy); Context (archaeology); Domain-specific language; Specification language; Complement (music); Medical diagnosis; SIGNAL (programming language); Programming language; Root cause; Theoretical computer science; Reliability engineering","score_opus":0.04247656139871854,"score_gpt":0.2590470647484599,"score_spread":0.21657050334974137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320005455","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0061204317,0.00018795376,0.98696005,0.00018571143,0.000059337013,0.00013005923,0.00022750346,0.0040418934,0.0020869724],"genre_scores_gemma":[0.3076414,0.00050052913,0.6856936,0.00044089786,0.000094538744,0.00035340953,0.0013027095,0.00095000456,0.003022862],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99650466,0.0007247011,0.00036859684,0.0006550177,0.0015080709,0.00023889584],"domain_scores_gemma":[0.9883471,0.006838471,0.0012927622,0.001670564,0.0016657356,0.00018543402],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002062981,0.0013952047,0.00057565543,0.00223441,0.0006384623,0.002147503,0.0016380294,0.0012730949,0.0046607475],"category_scores_gemma":[0.015397203,0.00060434407,0.0017907378,0.00093719177,0.0019695172,0.0037153352,0.0018457278,0.0025371742,0.0009964453],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058636814,0.00027990204,0.0070215696,0.0014298664,0.00018287606,0.002667886,0.001513951,0.14642411,0.08626888,0.40392116,0.008895095,0.34080833],"study_design_scores_gemma":[0.00007837001,0.00021598821,0.0005499779,0.00025959525,0.00009624556,0.0011436443,0.00016552939,0.7455248,0.074032396,0.14570202,0.032141764,0.000089678986],"about_ca_topic_score_codex":0.0030746975,"about_ca_topic_score_gemma":0.0030725962,"teacher_disagreement_score":0.0046607475,"about_ca_system_score_codex":0.0014728514,"about_ca_system_score_gemma":0.0021160638,"threshold_uncertainty_score":0.0155918},"labels":[],"label_agreement":null},{"id":"W4323530087","doi":"10.1109/tse.2023.3253700","title":"SLocator: Localizing the Origin of SQL Queries in Database-Backed Web Applications","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Database; Stored procedure; SQL; Query by Example; SQL injection; Programming language; Information retrieval; Web search query","score_opus":0.016398805950189235,"score_gpt":0.24398300974008116,"score_spread":0.22758420378989191,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323530087","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36863875,0.0016615121,0.22873384,0.00042373769,0.0001602496,0.0005007602,0.0048736343,0.38775826,0.007249241],"genre_scores_gemma":[0.7666295,0.00043053334,0.20933275,0.00056379195,0.000048600996,0.00020000867,0.013966892,0.004627626,0.004200336],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975414,0.00036689654,0.00023925114,0.00060526707,0.00097274565,0.00027448937],"domain_scores_gemma":[0.99474937,0.0015957812,0.0010011719,0.0009811709,0.0014553454,0.00021713358],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001637793,0.0013842653,0.0006131332,0.0038520868,0.00069000333,0.0015584091,0.001825008,0.0009567546,0.0015369143],"category_scores_gemma":[0.008270949,0.00067305745,0.0008883342,0.001616255,0.0006689605,0.002572518,0.0020463297,0.000983812,0.001574463],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025241175,0.00093542726,0.1603641,0.0016973133,0.00035605283,0.001481228,0.0026016128,0.027185163,0.12980582,0.00675267,0.068255015,0.59804153],"study_design_scores_gemma":[0.00022022809,0.0008882965,0.054858647,0.00026169358,0.00018447395,0.0012315804,0.0011063793,0.7041767,0.18356858,0.0092956675,0.043931544,0.0002762329],"about_ca_topic_score_codex":0.019372355,"about_ca_topic_score_gemma":0.021868631,"teacher_disagreement_score":0.019372355,"about_ca_system_score_codex":0.0009107119,"about_ca_system_score_gemma":0.0019209728,"threshold_uncertainty_score":0.038519144},"labels":[],"label_agreement":null},{"id":"W4324291633","doi":"10.1109/tse.2023.3256939","title":"New Techniques for Static Symmetry Breaking in Many-Sorted Finite Model Finding","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Machine Learning and Algorithms","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; University of British Columbia","funders":"","keywords":"Computer science; Symmetry breaking; Correctness; Theoretical computer science; sort; Symmetry (geometry); Context (archaeology); Inference; Algorithm; Mathematics; Artificial intelligence; Physics","score_opus":0.017929932504266763,"score_gpt":0.26386151031678046,"score_spread":0.24593157781251368,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4324291633","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006753418,0.000099835794,0.9895927,0.00033363508,0.000057119614,0.00011723013,0.00006412119,0.0013428329,0.0016391269],"genre_scores_gemma":[0.15573482,0.000156939,0.8412503,0.00034470568,0.00007149791,0.00018280331,0.00027066428,0.0007147834,0.0012734738],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.988915,0.0035677732,0.0011797182,0.0013577132,0.0040930356,0.0008867061],"domain_scores_gemma":[0.9704775,0.012385427,0.0017614812,0.012667615,0.002363222,0.00034473476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011598608,0.0014914925,0.0014533533,0.0026320019,0.0022719426,0.0050337748,0.005810152,0.0021560602,0.00610001],"category_scores_gemma":[0.041742943,0.0012953923,0.004307939,0.0027853705,0.006348185,0.012229588,0.009502246,0.0077649807,0.0012760814],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028460906,0.0002586282,0.0027117976,0.00043754236,0.00017915668,0.00024840195,0.00094949204,0.06777975,0.0128617175,0.68781054,0.003480092,0.22299829],"study_design_scores_gemma":[0.00010449035,0.0001566302,0.00031240756,0.00012532016,0.00014192579,0.00030551426,0.00030810034,0.29322386,0.03160539,0.6632533,0.010364449,0.00009854943],"about_ca_topic_score_codex":0.0022098823,"about_ca_topic_score_gemma":0.0038076346,"teacher_disagreement_score":0.011598608,"about_ca_system_score_codex":0.0031953766,"about_ca_system_score_gemma":0.0058022845,"threshold_uncertainty_score":0.061339974},"labels":[],"label_agreement":null},{"id":"W4361986063","doi":"10.1109/tse.2023.3256322","title":"Metamorphic Testing for Web System Security","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Computer science; Oracle; Security testing; Executable; Fuzz testing; Web application; Programming language; Database; Software engineering; Software; World Wide Web; Operating system; Cloud computing; Cloud computing security","score_opus":0.029534783520755,"score_gpt":0.24176754022297112,"score_spread":0.2122327567022161,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4361986063","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022335755,0.00056655006,0.96429014,0.00059374847,0.00004848357,0.00009443628,0.00006618935,0.0025661069,0.009438644],"genre_scores_gemma":[0.56003296,0.0010131899,0.43097922,0.0005649964,0.000119503755,0.00020841158,0.00032620915,0.0005162254,0.0062392764],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973865,0.0008588669,0.00021664938,0.00039384598,0.0009898535,0.0001543055],"domain_scores_gemma":[0.9965299,0.0021349702,0.00032399932,0.0006081317,0.00030687722,0.00009620995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015904034,0.00054745213,0.0004170223,0.0009019571,0.00036719377,0.0011076713,0.00070288574,0.0009227603,0.0028237135],"category_scores_gemma":[0.0056901625,0.00037931692,0.00088035036,0.00050059846,0.0019111462,0.0015981041,0.0014603386,0.0017137448,0.00041194135],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022769965,0.0002040968,0.00417879,0.00040491327,0.00006673832,0.0009551243,0.0005368879,0.11987886,0.043321118,0.5011209,0.0047520744,0.32435277],"study_design_scores_gemma":[0.00004955233,0.00020263049,0.0013414872,0.00024209188,0.000049912695,0.00090314855,0.000060194947,0.69492286,0.028759863,0.2542936,0.019127201,0.000047410627],"about_ca_topic_score_codex":0.0010867806,"about_ca_topic_score_gemma":0.0007908345,"teacher_disagreement_score":0.0028237135,"about_ca_system_score_codex":0.001012533,"about_ca_system_score_gemma":0.00071416487,"threshold_uncertainty_score":0.009446263},"labels":[],"label_agreement":null},{"id":"W4364321651","doi":"10.1109/tse.2023.3265855","title":"Identifying Concepts in Software Projects","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Documentation; Software engineering; Domain (mathematical analysis); Software project management; Software development; Software; Domain analysis; Data science; Software construction; Programming language","score_opus":0.03530785764332117,"score_gpt":0.2955445288280574,"score_spread":0.2602366711847362,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4364321651","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38433805,0.0027643899,0.5825015,0.0011791515,0.00018807875,0.00081111636,0.0067994236,0.0025884062,0.01882985],"genre_scores_gemma":[0.3974297,0.0011610824,0.58830225,0.00017157746,0.00004668442,0.0007902655,0.0087039005,0.00039060012,0.0030038897],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.995284,0.0012171986,0.00058513955,0.0011087258,0.0016118664,0.00019311512],"domain_scores_gemma":[0.9835096,0.008421813,0.0023099163,0.0014914752,0.003724908,0.0005422031],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032699876,0.0007465834,0.00040846676,0.0142196175,0.0015535436,0.0029866546,0.0010002928,0.0012665759,0.0022482718],"category_scores_gemma":[0.03015948,0.0004807557,0.0007231347,0.011668959,0.0010598444,0.008359853,0.004711513,0.001297858,0.0008022797],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023845003,0.0002323333,0.09286644,0.0014793681,0.000109467386,0.0010492381,0.01567258,0.0062195812,0.010596889,0.08616657,0.01172877,0.7736403],"study_design_scores_gemma":[0.00012900801,0.00039715122,0.10532776,0.002017922,0.00030640032,0.004391159,0.024560861,0.14678283,0.029036734,0.32280064,0.3639205,0.00032906773],"about_ca_topic_score_codex":0.00509015,"about_ca_topic_score_gemma":0.004294871,"teacher_disagreement_score":0.0142196175,"about_ca_system_score_codex":0.001341047,"about_ca_system_score_gemma":0.002744616,"threshold_uncertainty_score":0.017293513},"labels":[],"label_agreement":null},{"id":"W4367016230","doi":"10.1109/tse.2023.3269804","title":"A Search-Based Testing Approach for Deep Reinforcement Learning Agents","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Cisco Systems (Canada); École de Technologie Supérieure; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Machine learning; Software testing; Software engineering; Programming language; Software","score_opus":0.04544628432889011,"score_gpt":0.25729172732987793,"score_spread":0.21184544300098782,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367016230","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04858564,0.00032042782,0.9451978,0.0005923502,0.000056740773,0.00015587345,0.00006094163,0.0017000509,0.0033301448],"genre_scores_gemma":[0.8344012,0.00010196091,0.16252422,0.00034652906,0.000031115305,0.00023597223,0.00010533221,0.00013091254,0.002122773],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99802303,0.0007338246,0.0001201682,0.0003591236,0.00046430144,0.0002995104],"domain_scores_gemma":[0.9933891,0.004184209,0.00055726833,0.00046234977,0.0010059758,0.00040100692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027163944,0.0011532074,0.0013423503,0.000852665,0.00058863097,0.0010081884,0.0028740403,0.0017850004,0.0026132977],"category_scores_gemma":[0.011490425,0.00064077426,0.00082829257,0.0004088538,0.0020611372,0.0017438735,0.0018497461,0.0018082209,0.00036773641],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016019838,0.00014195862,0.0019195474,0.00008650802,0.000052684674,0.0001695513,0.000101372396,0.92903984,0.0024764563,0.017337697,0.0008314451,0.04768268],"study_design_scores_gemma":[0.000014720185,0.000038031856,0.000046356614,0.000004846043,0.00000452049,0.000011139421,0.0000060290254,0.99540675,0.00037653462,0.0039166776,0.00017091591,0.0000035906353],"about_ca_topic_score_codex":0.006458457,"about_ca_topic_score_gemma":0.0049098185,"teacher_disagreement_score":0.006458457,"about_ca_system_score_codex":0.0015694884,"about_ca_system_score_gemma":0.0026025602,"threshold_uncertainty_score":0.014365852},"labels":[],"label_agreement":null},{"id":"W4375928762","doi":"10.1109/tse.2023.3272631","title":"Discovering Reusable Functional Features in Legacy Object-Oriented Systems","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Concordia University; École de Technologie Supérieure; Collège de Bois-de-Boulogne","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Code refactoring; Programming language; Java; Object-oriented programming; Software design pattern; Reuse; Inheritance (genetic algorithm); Set (abstract data type); Delegation; Software engineering; Software","score_opus":0.0242138522624414,"score_gpt":0.2522181008092857,"score_spread":0.22800424854684428,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4375928762","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63402086,0.0005790223,0.35325226,0.0002840411,0.00002030921,0.0002645047,0.00072912977,0.008617404,0.002232514],"genre_scores_gemma":[0.5065929,0.00022122529,0.4886226,0.000053254313,0.0000113559745,0.000108907254,0.0024633098,0.0005615496,0.001364915],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99856925,0.00023364133,0.00013906171,0.00029173656,0.000646145,0.00012012142],"domain_scores_gemma":[0.9898756,0.0054099667,0.001484041,0.0014818849,0.0014815535,0.00026704956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016429893,0.00077434577,0.00047952708,0.0041828854,0.00085799914,0.0013421976,0.0013941786,0.0008522982,0.0005452358],"category_scores_gemma":[0.010138563,0.0007449108,0.0006990776,0.0024430044,0.00071700173,0.0019632166,0.0013832148,0.0006014288,0.0002865064],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037869744,0.0006727354,0.124641284,0.00085273053,0.00013299065,0.0020944215,0.0026489699,0.08037269,0.04287171,0.006362665,0.004276028,0.7346951],"study_design_scores_gemma":[0.00009160437,0.00027635702,0.05443724,0.00011589941,0.0001301988,0.0010783714,0.0009089406,0.8770453,0.039704047,0.016835876,0.009283984,0.00009221944],"about_ca_topic_score_codex":0.0072760466,"about_ca_topic_score_gemma":0.011510194,"teacher_disagreement_score":0.0072760466,"about_ca_system_score_codex":0.0010355287,"about_ca_system_score_gemma":0.0012281073,"threshold_uncertainty_score":0.014467418},"labels":[],"label_agreement":null},{"id":"W4379409284","doi":"10.1109/tse.2023.3282981","title":": A Semantics-Guided Safety Enhancement Framework for AI-Enabled Cyber-Physical Systems","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Safety Systems Engineering in Autonomy","field":"Engineering","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund","keywords":"Computer science; Artificial intelligence; Cyber-physical system; Robotics; Modular design; Flexibility (engineering); Redundancy (engineering); Machine learning; Robot; Programming language","score_opus":0.015201457747041808,"score_gpt":0.24747335626861416,"score_spread":0.23227189852157235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379409284","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009985149,0.00011801217,0.9947114,0.00022425357,0.00004451841,0.000088126864,0.00008289544,0.0012117341,0.0025205587],"genre_scores_gemma":[0.10630026,0.00049379986,0.8853371,0.00030197485,0.00013174723,0.00050412276,0.0005929816,0.00059897726,0.0057389946],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99847084,0.00042813263,0.0001362863,0.00024292462,0.0005269849,0.00019487327],"domain_scores_gemma":[0.9991646,0.00026756932,0.000097606346,0.00018438062,0.0001941432,0.000091782524],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029278495,0.0012848128,0.000802144,0.0012971665,0.0010422331,0.0029788862,0.0031163192,0.0016197877,0.004477302],"category_scores_gemma":[0.002613402,0.0006627161,0.0028334241,0.0007437241,0.0025585261,0.004378456,0.0047852285,0.002768383,0.0013721825],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000101016216,0.0000983391,0.00056175294,0.00037995903,0.00008194894,0.0004302951,0.00063709426,0.19165204,0.0050411406,0.7056954,0.0073242895,0.08799675],"study_design_scores_gemma":[0.000042325275,0.00008361795,0.00013496497,0.00012271146,0.000063012594,0.0001972193,0.00014055619,0.6537638,0.0039979112,0.28757676,0.05384143,0.00003564139],"about_ca_topic_score_codex":0.0050014374,"about_ca_topic_score_gemma":0.005830882,"teacher_disagreement_score":0.0050014374,"about_ca_system_score_codex":0.0014166947,"about_ca_system_score_gemma":0.0029456334,"threshold_uncertainty_score":0.015484154},"labels":[],"label_agreement":null},{"id":"W4380520359","doi":"10.1109/tse.2023.3285743","title":"STRE: An Automated Approach to Suggesting App Developers When to Stop Reading Reviews","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Fundo para o Desenvolvimento das Ciências e da Tecnologia; China Postdoctoral Science Foundation","keywords":"Computer science; Reading (process); Upload; Categorization; World Wide Web; Complement (music); Data science; Artificial intelligence","score_opus":0.039493449767354936,"score_gpt":0.2921104368037568,"score_spread":0.2526169870364019,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380520359","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14010514,0.0077937907,0.50420415,0.003042444,0.0018219817,0.005065976,0.019341573,0.3025094,0.016115516],"genre_scores_gemma":[0.25184172,0.001236503,0.70723677,0.0010152603,0.0005985876,0.0012413935,0.015792957,0.0026699118,0.018366931],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9941362,0.0013485475,0.0007134914,0.0016688505,0.0019033443,0.00022948305],"domain_scores_gemma":[0.9691749,0.012210709,0.005215107,0.0019490274,0.010370915,0.0010792708],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034935377,0.0034186451,0.0017693723,0.009945918,0.0010390042,0.0022635176,0.0026627306,0.001780393,0.004945308],"category_scores_gemma":[0.024979094,0.0009846794,0.0009640992,0.0030048115,0.00044285486,0.0027493904,0.0015379209,0.0015819488,0.0076220613],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009520223,0.0007526611,0.021572609,0.0020891505,0.00025556693,0.00074600155,0.001987172,0.003665518,0.023949549,0.0010491249,0.11443197,0.8285486],"study_design_scores_gemma":[0.00087763363,0.0029002998,0.063588046,0.000899997,0.0011068981,0.0031590245,0.003284667,0.5872862,0.06736445,0.00677787,0.26188198,0.0008730623],"about_ca_topic_score_codex":0.009737097,"about_ca_topic_score_gemma":0.030253671,"teacher_disagreement_score":0.009945918,"about_ca_system_score_codex":0.0009383641,"about_ca_system_score_gemma":0.003422146,"threshold_uncertainty_score":0.01936084},"labels":[],"label_agreement":null},{"id":"W4381304075","doi":"10.1109/tse.2023.3281275","title":"Multi-Granularity Detector for Vulnerability Fixes","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"National Research Foundation Singapore; National University of Singapore","keywords":"Computer science; Commit; Granularity; Vulnerability (computing); Source code; Software; Python (programming language); Code (set theory); Data mining; Computer security; Database; Operating system; Programming language","score_opus":0.03505384984562248,"score_gpt":0.28615123860887126,"score_spread":0.2510973887632488,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4381304075","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42391136,0.006564897,0.50737244,0.0015563077,0.0010635031,0.0005183207,0.0072880685,0.04458587,0.007139233],"genre_scores_gemma":[0.8544989,0.00074189855,0.13024047,0.00036997432,0.00016422475,0.0001907835,0.008783074,0.00046041858,0.004550248],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9968094,0.00030482645,0.00021214846,0.0011081104,0.0012639454,0.00030152086],"domain_scores_gemma":[0.9939466,0.0021271787,0.0009897514,0.0010015774,0.0015776946,0.00035731943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002675482,0.001494858,0.001298812,0.0082818335,0.00068042916,0.0012452536,0.001928968,0.0017014424,0.0013403945],"category_scores_gemma":[0.010378172,0.000361568,0.0010970378,0.0026930226,0.0005742216,0.0025449235,0.002341325,0.0024117862,0.0012731053],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007344626,0.0006798523,0.12889387,0.0006094997,0.00051970774,0.001105731,0.00037950862,0.07732664,0.028409386,0.0032137127,0.042690378,0.71543735],"study_design_scores_gemma":[0.000034862464,0.0002858284,0.026829813,0.00009263248,0.00015462255,0.0011097324,0.00019232226,0.932365,0.023232246,0.004829755,0.010800124,0.00007305984],"about_ca_topic_score_codex":0.0045066513,"about_ca_topic_score_gemma":0.0070486674,"teacher_disagreement_score":0.0082818335,"about_ca_system_score_codex":0.0009933676,"about_ca_system_score_gemma":0.0012109684,"threshold_uncertainty_score":0.014149487},"labels":[],"label_agreement":null},{"id":"W4382203557","doi":"10.1109/tse.2023.3289808","title":"Self-Admitted Technical Debt in Ethereum Smart Contracts: A Large-Scale Exploratory Study","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Technical debt; Computer science; Workaround; Context (archaeology); Code (set theory); Implementation; Data science; Software engineering; Software development; Programming language; Software","score_opus":0.009701457661906625,"score_gpt":0.2289408710523694,"score_spread":0.21923941339046277,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382203557","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99512357,0.0001890499,0.0020909547,0.00025962526,0.0000042202305,0.000109143766,0.00037233182,0.000020107882,0.0018308506],"genre_scores_gemma":[0.9944218,0.0002262228,0.0030846242,0.00015135409,0.000009770882,0.0002038775,0.00078314723,0.00003572268,0.0010836277],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9886846,0.0057593305,0.00074544846,0.00083133194,0.0031709778,0.00080820953],"domain_scores_gemma":[0.8547771,0.10811609,0.021155803,0.0061139036,0.0078466805,0.0019903218],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013577002,0.000291534,0.0003829021,0.003940326,0.001612993,0.0022029046,0.0011046407,0.0011180014,0.00191367],"category_scores_gemma":[0.06309479,0.00041680687,0.00037905158,0.004521567,0.002413594,0.00478376,0.0029534483,0.0018142721,0.0005646621],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000374355,0.0014984463,0.72252,0.0012117547,0.000110663306,0.0044181375,0.15985768,0.0021110312,0.0043129893,0.0136769125,0.006131999,0.08377606],"study_design_scores_gemma":[0.00005917586,0.00051786157,0.69083786,0.001197531,0.00007134222,0.0022288237,0.23271836,0.018074172,0.0033578204,0.0059323087,0.044855643,0.00014911439],"about_ca_topic_score_codex":0.0039039166,"about_ca_topic_score_gemma":0.005776921,"teacher_disagreement_score":0.013577002,"about_ca_system_score_codex":0.0018785275,"about_ca_system_score_gemma":0.00225109,"threshold_uncertainty_score":0.071802914},"labels":[],"label_agreement":null},{"id":"W4382318094","doi":"10.1109/tse.2023.3288901","title":"NLP-Based Automated Compliance Checking of Data Processing Agreements Against GDPR","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Access Control and Trust","field":"Social Sciences","cited_by":77,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg","keywords":"Computer science; Table (database); General Data Protection Regulation; Ambiguity; Software; Glossary; Compliance (psychology); Information retrieval; Software engineering; Database; Data Protection Act 1998; Computer security; Programming language","score_opus":0.09377241932177718,"score_gpt":0.3490663702075247,"score_spread":0.25529395088574747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382318094","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053389926,0.00018053647,0.89092076,0.0018703451,0.0001726253,0.0019650958,0.0037643076,0.039843526,0.007892913],"genre_scores_gemma":[0.15008247,0.00015148324,0.8365605,0.00047591684,0.00006759506,0.0007718321,0.007583139,0.0018654697,0.002441563],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.96151185,0.016280388,0.0047774278,0.0066951397,0.009764098,0.0009711451],"domain_scores_gemma":[0.8852486,0.07465058,0.009264592,0.015079005,0.014984478,0.0007727668],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017674342,0.001239054,0.0013820127,0.0058911135,0.002620095,0.0051389746,0.0029332864,0.0024218282,0.006724085],"category_scores_gemma":[0.08601592,0.0012228003,0.001991699,0.002212725,0.0024729867,0.0063063274,0.0055210297,0.0032826383,0.0038320068],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007971698,0.0010202668,0.011727498,0.0029416797,0.00031386086,0.006026465,0.010949708,0.08702513,0.094040394,0.06709025,0.04730982,0.6707577],"study_design_scores_gemma":[0.0002686368,0.00018024382,0.0046119937,0.0004774353,0.00016196986,0.0012899375,0.002764345,0.75699985,0.11007484,0.042263057,0.08065115,0.0002566202],"about_ca_topic_score_codex":0.011026906,"about_ca_topic_score_gemma":0.009557454,"teacher_disagreement_score":0.017674342,"about_ca_system_score_codex":0.002992193,"about_ca_system_score_gemma":0.008179484,"threshold_uncertainty_score":0.093471944},"labels":[],"label_agreement":null},{"id":"W4386212357","doi":"10.1109/tse.2023.3307243","title":"ADPTriage: Approximate Dynamic Programming for Bug Triage","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Toronto Metropolitan University","funders":"","keywords":"Computer science; Triage; Markov decision process; Process (computing); Software bug; Software regression; Task (project management); Software; Pipeline (software); Software engineering; Markov process; Software development; Programming language; Software quality; Systems engineering","score_opus":0.019948979039290816,"score_gpt":0.2724453377257305,"score_spread":0.25249635868643966,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386212357","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0073207873,0.00044640462,0.9883762,0.000390636,0.00008573936,0.00009397589,0.00015545837,0.00046018467,0.0026707042],"genre_scores_gemma":[0.55207026,0.00081823004,0.43921295,0.0006480045,0.00015679646,0.0006833116,0.00068432145,0.00038839743,0.005337838],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99822646,0.00071486953,0.000089521905,0.00036009747,0.00035258775,0.0002564558],"domain_scores_gemma":[0.99246055,0.0060836836,0.0004380298,0.00025402717,0.00046873244,0.00029504814],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032645494,0.0017516279,0.0025929702,0.00094496383,0.0008078345,0.0018758712,0.0022433184,0.0020608294,0.0052596545],"category_scores_gemma":[0.010986083,0.0013019156,0.0013389329,0.0013988061,0.001114726,0.0016451816,0.002254685,0.0035471069,0.000652707],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000058081616,0.000048138452,0.0003415416,0.000086630374,0.00003102455,0.000034568326,0.000036212066,0.9724663,0.00017258551,0.0075316182,0.0013441826,0.01784909],"study_design_scores_gemma":[0.0000074298578,0.000013004884,0.000022373195,0.000005385126,0.0000038232974,0.0000049844043,0.0000048710303,0.9955852,0.00003802464,0.0040328945,0.00027937375,0.0000025328939],"about_ca_topic_score_codex":0.012092539,"about_ca_topic_score_gemma":0.010855676,"teacher_disagreement_score":0.012092539,"about_ca_system_score_codex":0.0018333024,"about_ca_system_score_gemma":0.0040429477,"threshold_uncertainty_score":0.024044275},"labels":[],"label_agreement":null},{"id":"W4386825342","doi":"10.1109/tse.2023.3313875","title":"A Grounded Theory of Cross-Community SECOs: Feedback Diversity Versus Synchronization","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Grounded theory; Computer science; Synchronization (alternating current); Diversity (politics); Terabyte; Upstream (networking); Data science; World Wide Web; Operating system; Qualitative research; Computer network; Sociology","score_opus":0.039422114663781684,"score_gpt":0.27669226271369296,"score_spread":0.23727014804991128,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386825342","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17134169,0.0006244581,0.7493337,0.014226176,0.00013803136,0.0022445282,0.00048156182,0.00017265904,0.06143719],"genre_scores_gemma":[0.86428636,0.0002792364,0.13191652,0.0007054314,0.000029041821,0.0017489049,0.00018537563,0.000034552013,0.0008145567],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9703946,0.021707471,0.0011837371,0.0027696083,0.0032248888,0.0007196949],"domain_scores_gemma":[0.9379835,0.047927644,0.0039577098,0.0046978514,0.00397342,0.0014598309],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.030055461,0.0010425036,0.000863821,0.0057804463,0.0047545996,0.006673624,0.0028612756,0.0024294804,0.0022762138],"category_scores_gemma":[0.037267424,0.0008789523,0.0010693744,0.0041541797,0.0315471,0.011852937,0.0049065254,0.0032494983,0.00030092348],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057966532,0.00016302294,0.008372629,0.0004904235,0.000049677383,0.00020410758,0.1029215,0.0032709981,0.00089340156,0.8567798,0.0008783828,0.025918063],"study_design_scores_gemma":[0.00014770955,0.000152517,0.0049031028,0.0009029751,0.000054369677,0.00020520958,0.06579667,0.0203771,0.0009954888,0.8819536,0.024445688,0.00006563278],"about_ca_topic_score_codex":0.006172457,"about_ca_topic_score_gemma":0.005535624,"teacher_disagreement_score":0.030055461,"about_ca_system_score_codex":0.012740826,"about_ca_system_score_gemma":0.008677708,"threshold_uncertainty_score":0.15895039},"labels":[],"label_agreement":null},{"id":"W4388469715","doi":"10.1109/tse.2023.3327575","title":"Identifying the Hazard Boundary of ML-Enabled Autonomous Systems Using Cooperative Coevolutionary Search","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Mitacs; Canada Research Chairs","keywords":"Computer science; Boundary (topology); Hazard; Context (archaeology); Component (thermodynamics); Genetic algorithm; Artificial intelligence; Metaheuristic; Machine learning; Algorithm; Mathematics","score_opus":0.03903614166885182,"score_gpt":0.2854306910932624,"score_spread":0.24639454942441058,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388469715","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34571385,0.00038400487,0.64707094,0.00052724895,0.000022832297,0.000163128,0.000058596845,0.00031801,0.0057412405],"genre_scores_gemma":[0.9444811,0.00010021695,0.05408218,0.00008268487,0.000009095401,0.00016096374,0.00006909133,0.00003206045,0.0009825755],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994721,0.00017865785,0.000031253283,0.00011540475,0.00009621869,0.0001064319],"domain_scores_gemma":[0.9957967,0.0030205983,0.0004439433,0.00016962606,0.00029078312,0.00027830678],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019521809,0.00095079665,0.0011465613,0.0012008717,0.0007832846,0.001441831,0.001502782,0.0016147693,0.0016023932],"category_scores_gemma":[0.0075767036,0.0008700341,0.00096451864,0.00045438198,0.0018176491,0.0015087762,0.0025957564,0.0014417616,0.0001680556],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046485005,0.000035077144,0.0025688836,0.000035657264,0.000048526217,0.00009779306,0.00012187651,0.9831872,0.000815236,0.006182905,0.00016072458,0.006699688],"study_design_scores_gemma":[0.000006657847,0.000021863074,0.00017316481,0.000006974909,0.000008163186,0.000010680188,0.000035497294,0.9943224,0.00020828067,0.0050908127,0.00011112583,0.0000044201684],"about_ca_topic_score_codex":0.0065249004,"about_ca_topic_score_gemma":0.003861799,"teacher_disagreement_score":0.0065249004,"about_ca_system_score_codex":0.0014485306,"about_ca_system_score_gemma":0.0018553972,"threshold_uncertainty_score":0.012973845},"labels":[],"label_agreement":null},{"id":"W4388623088","doi":"10.1109/tse.2023.3331254","title":"Concretization of Abstract Traffic Scene Specifications Using Metaheuristic Search","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Autonomous Vehicle Technology and Safety","field":"Engineering","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Nemzeti Kutatási, Fejlesztési és Innovaciós Alap; Natural Sciences and Engineering Research Council of Canada; National Research, Development and Innovation Office; Innovációs és Technológiai Minisztérium; Nemzeti Kutatási Fejlesztési és Innovációs Hivatal","keywords":"Computer science; Context (archaeology); Scalability; Constraint (computer-aided design); Metaheuristic; Set (abstract data type); Real-time computing; Artificial intelligence; Programming language; Database","score_opus":0.04044585752119628,"score_gpt":0.23861477820355717,"score_spread":0.1981689206823609,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388623088","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09028085,0.00008279387,0.9026085,0.00015317381,0.000023723875,0.00030012065,0.00029729927,0.0023337293,0.003919827],"genre_scores_gemma":[0.41236773,0.00009804481,0.5834281,0.00014941339,0.00001046793,0.00033026378,0.0015569924,0.00063025544,0.0014287962],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990767,0.00035333296,0.00005069654,0.00013343696,0.00027357406,0.00011220938],"domain_scores_gemma":[0.99803,0.0011548077,0.00021402435,0.00026811252,0.0002731788,0.000059844366],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011825884,0.0014540041,0.0006717239,0.0011678032,0.0004188827,0.001031398,0.0014283533,0.00092860206,0.002357051],"category_scores_gemma":[0.0041245287,0.0006723938,0.00146429,0.00062676053,0.001007382,0.0009917034,0.0016807866,0.0011096891,0.00028533643],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005879327,0.000097081465,0.0009299529,0.00010445092,0.00003947529,0.00013372312,0.00010512858,0.9527079,0.0060288254,0.0070866,0.00078026904,0.03192785],"study_design_scores_gemma":[0.000018104207,0.000037284033,0.00010539441,0.000010330234,0.0000089040295,0.000020568528,0.00006412276,0.9938438,0.0023997175,0.0026399246,0.00084557344,0.000006280775],"about_ca_topic_score_codex":0.005948362,"about_ca_topic_score_gemma":0.009133617,"teacher_disagreement_score":0.005948362,"about_ca_system_score_codex":0.0010843968,"about_ca_system_score_gemma":0.0018314744,"threshold_uncertainty_score":0.011827469},"labels":[],"label_agreement":null},{"id":"W4388758252","doi":"10.1109/tse.2023.3332568","title":"Properties and Styles of Software Technology Tutorials","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; JavaScript; Python (programming language); Documentation; Software; TypeScript; Java; Resource (disambiguation); World Wide Web; Software engineering; Information retrieval; Programming language","score_opus":0.025863080000366374,"score_gpt":0.23587883220421202,"score_spread":0.21001575220384563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388758252","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8974189,0.00093420746,0.06918813,0.0002526372,0.00003527754,0.00026182225,0.003456474,0.0019727591,0.026479706],"genre_scores_gemma":[0.9694569,0.00025626103,0.023564598,0.000047350524,0.000036187692,0.0001960281,0.0032095013,0.00047472882,0.0027584508],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.98853475,0.0041256538,0.0016049455,0.0015334359,0.0035137169,0.0006875992],"domain_scores_gemma":[0.87407136,0.067504674,0.02351988,0.012456725,0.017671041,0.004776269],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0058185006,0.00046447155,0.0004356266,0.008416725,0.00095720816,0.004696299,0.00087617955,0.000719772,0.0031700053],"category_scores_gemma":[0.10379945,0.0005593051,0.00067100726,0.0066967797,0.0010552256,0.0054889387,0.0017192855,0.00079257536,0.0009182045],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009242467,0.0004120711,0.5665036,0.0011407346,0.00030914365,0.00084656634,0.014383565,0.01523705,0.021428967,0.08146887,0.0062105427,0.2911347],"study_design_scores_gemma":[0.0001271482,0.0009431941,0.6772908,0.00064046634,0.00033930043,0.0036040824,0.008518594,0.08952387,0.030250786,0.08725972,0.10110138,0.0004006696],"about_ca_topic_score_codex":0.0016057687,"about_ca_topic_score_gemma":0.0015466028,"teacher_disagreement_score":0.008416725,"about_ca_system_score_codex":0.0015556797,"about_ca_system_score_gemma":0.0010635625,"threshold_uncertainty_score":0.030771554},"labels":[],"label_agreement":null},{"id":"W4391164126","doi":"10.1109/tse.2024.3358258","title":"Multi-Language Software Development: Issues, Challenges, and Solutions","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Office of Naval Research","keywords":"Computer science; Interoperability; World Wide Web; Software development; Software; Software engineering; Programming language; Data science","score_opus":0.03412181408304442,"score_gpt":0.26704306635803715,"score_spread":0.23292125227499272,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391164126","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.48844504,0.04794777,0.08206375,0.3463203,0.0017764547,0.00044683454,0.00033447138,0.0013531366,0.031312186],"genre_scores_gemma":[0.8720332,0.02241936,0.084394455,0.009981507,0.0015052917,0.00036875517,0.00042975848,0.00049948605,0.0083682],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9692848,0.015427383,0.002109468,0.0027778733,0.007937842,0.0024626993],"domain_scores_gemma":[0.8963726,0.068931736,0.010060695,0.0037398448,0.014714793,0.006180368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026497122,0.00077042496,0.00055551913,0.0050166207,0.0065710032,0.0108392015,0.0027634548,0.0035716933,0.002402509],"category_scores_gemma":[0.05956748,0.0009828911,0.00075439236,0.0063553387,0.0059029735,0.020661734,0.008221827,0.0046926783,0.0008730162],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015364806,0.0005631735,0.08236866,0.0033042473,0.000083000035,0.0034753967,0.13550325,0.0016959131,0.0073132208,0.05162569,0.032668024,0.6812458],"study_design_scores_gemma":[0.000052511667,0.00042131782,0.0644218,0.0036629206,0.00009867783,0.0070894645,0.505407,0.014295315,0.0053298334,0.094018824,0.30479047,0.00041182776],"about_ca_topic_score_codex":0.005019992,"about_ca_topic_score_gemma":0.008553477,"teacher_disagreement_score":0.026497122,"about_ca_system_score_codex":0.0039890525,"about_ca_system_score_gemma":0.00853379,"threshold_uncertainty_score":0.14013189},"labels":[],"label_agreement":null},{"id":"W4391164257","doi":"10.1109/tse.2024.3358283","title":"Tracking the Evolution of Static Code Warnings: The State-of-the-Art and a Better Approach","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Static analysis; Static program analysis; Code (set theory); Workflow; Tracking (education); Source code; Software engineering; Software; Tracking system; Software evolution; Code smell; Programming language; Software development; Artificial intelligence; Software quality; Database; Software construction","score_opus":0.011706816219258082,"score_gpt":0.22705450869958402,"score_spread":0.21534769248032593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391164257","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51566744,0.050157752,0.34230635,0.010305639,0.0018554458,0.0005777822,0.020328576,0.05375492,0.0050461534],"genre_scores_gemma":[0.61111146,0.0077378056,0.33123764,0.0016252139,0.00066558406,0.00040362866,0.040725075,0.0027338983,0.003759638],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9867621,0.0025923366,0.0015778723,0.004425033,0.0039596627,0.0006828855],"domain_scores_gemma":[0.9419053,0.024331434,0.0074346154,0.013987126,0.010704978,0.0016363816],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0075858305,0.0029000607,0.0024506582,0.014825568,0.0016573872,0.004984097,0.0041628825,0.003548477,0.0009105453],"category_scores_gemma":[0.045119207,0.0013068726,0.0020152503,0.00989883,0.0014553602,0.0077514243,0.0032565317,0.004069152,0.0011963693],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006761364,0.00075125083,0.20980676,0.0031569896,0.0007718878,0.0005244736,0.0017905574,0.02175712,0.015652578,0.0035150228,0.03995555,0.7016417],"study_design_scores_gemma":[0.0003058667,0.00096752256,0.17309283,0.0015234207,0.0011902315,0.0021000372,0.002961935,0.64536315,0.028385403,0.019153725,0.124377295,0.0005786522],"about_ca_topic_score_codex":0.023635311,"about_ca_topic_score_gemma":0.030410897,"teacher_disagreement_score":0.023635311,"about_ca_system_score_codex":0.0013403613,"about_ca_system_score_gemma":0.0036255557,"threshold_uncertainty_score":0.04699546},"labels":[],"label_agreement":null},{"id":"W4391559702","doi":"10.1109/tse.2024.3363223","title":"DynAMICS: A Tool-Based Method for the Specification and Dynamic Detection of Android Behavioral Code Smells","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code smell; Computer science; Android (operating system); Source code; Software; Code (set theory); Software quality; Programming language; Artificial intelligence; Software engineering; Software development; Operating system","score_opus":0.01749395245963264,"score_gpt":0.29060479010718,"score_spread":0.27311083764754734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391559702","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016527683,0.000068964786,0.8620592,0.00013426504,0.000098668665,0.0004782916,0.0010765805,0.13200876,0.0024225938],"genre_scores_gemma":[0.03772072,0.00025222445,0.9157574,0.00034084104,0.00008638324,0.0023811802,0.003806286,0.028512849,0.011142113],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99558216,0.0007017394,0.00057117187,0.0009919972,0.0018769763,0.00027590198],"domain_scores_gemma":[0.99163955,0.004374263,0.00081732584,0.0015381862,0.001311829,0.00031889515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036611003,0.0036963602,0.0014463842,0.0042786594,0.00095170294,0.0032398615,0.0028842664,0.0025126452,0.015251922],"category_scores_gemma":[0.01655347,0.002184758,0.0026538824,0.0010223867,0.0016397358,0.0041125044,0.004640372,0.002968779,0.010305689],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011403526,0.0005676309,0.011449877,0.002269701,0.00043227826,0.0022786693,0.0036621806,0.010582208,0.095589496,0.050983593,0.12337408,0.6976699],"study_design_scores_gemma":[0.0006552626,0.0004670155,0.0057257684,0.00080203486,0.00028032355,0.002553538,0.0007723587,0.3009022,0.14638427,0.033086907,0.507548,0.00082236127],"about_ca_topic_score_codex":0.0031696847,"about_ca_topic_score_gemma":0.004216821,"teacher_disagreement_score":0.015251922,"about_ca_system_score_codex":0.0010398964,"about_ca_system_score_gemma":0.0034405622,"threshold_uncertainty_score":0.051022828},"labels":[],"label_agreement":null},{"id":"W4391759551","doi":"10.1109/tse.2024.3362921","title":"Measuring and Characterizing (Mis)compliance of the Android Permission System","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Permission; Computer science; Android (operating system); Computer security; Operating system; Software engineering","score_opus":0.02234690032524389,"score_gpt":0.22126373324136642,"score_spread":0.19891683291612253,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391759551","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98229825,0.00022053534,0.013516586,0.00017160378,0.000027461892,0.00014408406,0.0003840622,0.00058003503,0.0026572836],"genre_scores_gemma":[0.9894671,0.00007310261,0.009357155,0.000038215872,0.000010198507,0.000093215815,0.0004930529,0.00009815271,0.0003698854],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.96465874,0.009085137,0.005577212,0.0042211036,0.015302782,0.0011550577],"domain_scores_gemma":[0.7559702,0.13100998,0.042504527,0.03151336,0.03732624,0.0016757098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014695042,0.000547876,0.000425648,0.004332884,0.0008545699,0.002331198,0.0011346466,0.0014410547,0.00046443523],"category_scores_gemma":[0.17701638,0.00067289517,0.00037790873,0.0028161258,0.0015058288,0.0041488274,0.001971024,0.001815427,0.0003699954],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031221748,0.00024100016,0.8932805,0.000341828,0.0001742073,0.00042546215,0.0068579623,0.011397762,0.013043668,0.0041969107,0.0011591307,0.06856933],"study_design_scores_gemma":[0.00003701839,0.0008044631,0.7514769,0.00021204878,0.00018196156,0.0023828351,0.0047870832,0.1990697,0.02863135,0.0049539744,0.007263862,0.00019892537],"about_ca_topic_score_codex":0.005615572,"about_ca_topic_score_gemma":0.0055083493,"teacher_disagreement_score":0.014695042,"about_ca_system_score_codex":0.0011736053,"about_ca_system_score_gemma":0.0020289777,"threshold_uncertainty_score":0.077715755},"labels":[],"label_agreement":null},{"id":"W4391974543","doi":"10.1109/tse.2024.3368208","title":"Software Testing With Large Language Models: Survey, Landscape, and Vision","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":394,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Software engineering; Software","score_opus":0.017120406674413505,"score_gpt":0.2519216021540697,"score_spread":0.23480119547965622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391974543","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035197347,0.45945078,0.44265267,0.03379083,0.00069568213,0.00033082167,0.000978464,0.00620898,0.02069431],"genre_scores_gemma":[0.42553413,0.33147672,0.22092924,0.008213248,0.0031031775,0.00058278657,0.0036683679,0.0018988195,0.004593555],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9900885,0.004911423,0.0006522069,0.0011580909,0.002961443,0.00022828947],"domain_scores_gemma":[0.9138474,0.07639063,0.0020218669,0.0029828313,0.0042288885,0.0005283364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009385707,0.0014915407,0.0018486273,0.0059925504,0.00043090034,0.0044823247,0.003635035,0.0021417132,0.0023850368],"category_scores_gemma":[0.06181108,0.0010074903,0.0015199229,0.005038078,0.003055187,0.010523414,0.0024927347,0.003093329,0.0011034748],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014588295,0.00020581722,0.010468917,0.003307177,0.00017093924,0.00019619561,0.0005721956,0.016502919,0.0010086969,0.02595612,0.013219993,0.9282451],"study_design_scores_gemma":[0.00011815417,0.00087011606,0.014190343,0.00932717,0.00046327608,0.0029726354,0.002998374,0.48164672,0.008588703,0.18447734,0.29400706,0.00034007456],"about_ca_topic_score_codex":0.0063318913,"about_ca_topic_score_gemma":0.0043580844,"teacher_disagreement_score":0.009385707,"about_ca_system_score_codex":0.002108319,"about_ca_system_score_gemma":0.0029164501,"threshold_uncertainty_score":0.04963696},"labels":[],"label_agreement":null},{"id":"W4392121835","doi":"10.1109/tse.2024.3366753","title":"Factoring Expertise, Workload, and Turnover Into Code Review Recommendation","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Factoring; Workload; Code (set theory); Programming language; Software engineering; Parallel computing; Operating system; Accounting","score_opus":0.0172653523531208,"score_gpt":0.2706072422291304,"score_spread":0.2533418898760096,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392121835","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.794895,0.0013909781,0.19479097,0.0012509048,0.00009328496,0.00041912732,0.0003790116,0.002244378,0.004536359],"genre_scores_gemma":[0.94429237,0.00024019976,0.05353475,0.000118706506,0.000052123058,0.0000839999,0.0002998986,0.000053496646,0.0013244898],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9967872,0.0013078518,0.00029267318,0.00061242207,0.0007482172,0.0002516594],"domain_scores_gemma":[0.9589378,0.027813392,0.0037328876,0.0025269415,0.005430045,0.0015588559],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056106113,0.00056875666,0.0007025098,0.001294548,0.0006214325,0.001691366,0.00100193,0.0011500583,0.001179955],"category_scores_gemma":[0.039110888,0.00056616944,0.00043185623,0.0009459051,0.00035139875,0.0023915533,0.00062765356,0.0010281844,0.0005481385],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009121501,0.0010766534,0.36251715,0.00052113354,0.0003805703,0.00036838645,0.0012175445,0.19851466,0.018102655,0.0015636354,0.0058590886,0.4089664],"study_design_scores_gemma":[0.00007020588,0.0007645437,0.056689847,0.00006834205,0.00017325707,0.00032964718,0.00035445957,0.93235505,0.0051540374,0.0012768586,0.002690026,0.000073753785],"about_ca_topic_score_codex":0.014129207,"about_ca_topic_score_gemma":0.03033427,"teacher_disagreement_score":0.014129207,"about_ca_system_score_codex":0.0012067356,"about_ca_system_score_gemma":0.0018118549,"threshold_uncertainty_score":0.029672146},"labels":[],"label_agreement":null},{"id":"W4394711492","doi":"10.1109/tse.2024.3387840","title":"Characterizing Timeout Builds in Continuous Integration","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Business Process Modeling and Analysis","field":"Business, Management and Accounting","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Timeout; Computer science; Programming language; Distributed computing; Computer network","score_opus":0.010545663966579999,"score_gpt":0.20854615975799312,"score_spread":0.19800049579141313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394711492","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9898768,0.0004131239,0.0063864295,0.0001672627,0.000016131544,0.000039527735,0.0020368006,0.00018259688,0.0008812524],"genre_scores_gemma":[0.99205244,0.00013064453,0.0036223193,0.000032423595,0.000023351702,0.000058075533,0.0037904151,0.000041860887,0.00024844048],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99329114,0.0015361952,0.0008771578,0.0020822375,0.0015161855,0.0006970115],"domain_scores_gemma":[0.87775356,0.07453385,0.029088177,0.008714084,0.0066593057,0.003251123],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008884896,0.0005901435,0.00059099455,0.0055005136,0.0006451289,0.002958959,0.0013302591,0.0010878562,0.0010560845],"category_scores_gemma":[0.058128987,0.00053579226,0.0007975514,0.0059994967,0.00095161493,0.0032672335,0.002185905,0.0012805504,0.00057846156],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008505896,0.00006263941,0.98397726,0.00005461542,0.000056977235,0.000100123165,0.00056108594,0.004060538,0.0002768457,0.00027519473,0.0006077507,0.009882005],"study_design_scores_gemma":[0.0000144015075,0.00014474188,0.9272444,0.00007383251,0.000060118786,0.0004121406,0.0014586431,0.064818434,0.0007837346,0.001341965,0.0035995052,0.000048070677],"about_ca_topic_score_codex":0.010414047,"about_ca_topic_score_gemma":0.014810053,"teacher_disagreement_score":0.010414047,"about_ca_system_score_codex":0.0010596304,"about_ca_system_score_gemma":0.000989533,"threshold_uncertainty_score":0.046988368},"labels":[],"label_agreement":null},{"id":"W4398187606","doi":"10.1109/tse.2024.3402157","title":"A Lean Simulation Framework for Stress Testing IoT Cloud Systems","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Digital Transformation in Industry","field":"Engineering","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Stem Cell Network; University of Ottawa","funders":"","keywords":"Computer science; Cloud computing; Stress testing (software); Cloud testing; Internet of Things; Software engineering; Distributed computing; Embedded system; Operating system; Cloud computing security","score_opus":0.03102711421944842,"score_gpt":0.2550498836915008,"score_spread":0.22402276947205235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398187606","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007432099,0.00005105569,0.9886623,0.00011283189,0.00001802178,0.00012772855,0.0001179398,0.0014273018,0.0020506128],"genre_scores_gemma":[0.26592016,0.00030985606,0.72964954,0.00012378833,0.000022953089,0.0007629502,0.00069082604,0.0005015766,0.002018422],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989587,0.00036929845,0.000096399934,0.00009376626,0.00037428003,0.000107456704],"domain_scores_gemma":[0.99869776,0.0006204364,0.00011981239,0.0001894174,0.00028148785,0.00009116558],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018733767,0.00096499163,0.00062026386,0.00086185563,0.0006032199,0.0013652763,0.0020051377,0.00080716546,0.0019050292],"category_scores_gemma":[0.0035859887,0.0004995417,0.001201065,0.0004875912,0.0012666036,0.0012911009,0.0016386353,0.001327574,0.00041470255],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028513228,0.000053917494,0.00069972395,0.000057910547,0.00001697476,0.000056045224,0.00010357032,0.95181173,0.002647663,0.03602537,0.0004950541,0.008003524],"study_design_scores_gemma":[0.000014425306,0.000021398866,0.00006291372,0.000016361548,0.0000066025286,0.000013463725,0.000019382587,0.9865264,0.0014023257,0.008889586,0.0030195734,0.0000075190187],"about_ca_topic_score_codex":0.010468997,"about_ca_topic_score_gemma":0.00852437,"teacher_disagreement_score":0.010468997,"about_ca_system_score_codex":0.0015590918,"about_ca_system_score_gemma":0.003127272,"threshold_uncertainty_score":0.020816088},"labels":[],"label_agreement":null},{"id":"W4399768519","doi":"10.1109/tse.2024.3411928","title":"LUNA: A Model-Based Universal Analysis Framework for Large Language Models","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Japan Society for the Promotion of Science; JST-Mirai Program; Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Programming language; Modeling language; Software engineering; Data science; Natural language processing; Software","score_opus":0.013074621913195335,"score_gpt":0.2445529802957176,"score_spread":0.23147835838252226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399768519","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00039830257,0.00013828327,0.9952748,0.0001397854,0.000019545334,0.00004092865,0.00023599039,0.0030244747,0.00072789704],"genre_scores_gemma":[0.06763932,0.0006280428,0.9226523,0.00039112303,0.00016772332,0.0006191774,0.0019837357,0.0023873148,0.0035311794],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99726164,0.001153142,0.0002281109,0.00044041077,0.0007125275,0.00020411565],"domain_scores_gemma":[0.99730885,0.001543475,0.00024575205,0.00042709944,0.00035655405,0.000118271135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003891363,0.0017338862,0.0013570401,0.002961291,0.0012126208,0.0035921584,0.0038001502,0.0014179994,0.008856321],"category_scores_gemma":[0.0098814415,0.0015350868,0.0049431426,0.0015380285,0.0018064566,0.0054466887,0.004421318,0.0040963735,0.0036351888],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012202023,0.00007548814,0.001037729,0.00038227357,0.0003087228,0.00040863242,0.0005345108,0.1461991,0.0025208506,0.7089982,0.016222944,0.1231895],"study_design_scores_gemma":[0.000015764632,0.000020074725,0.00008824165,0.00004928518,0.00003949091,0.00006975132,0.000040444997,0.73667455,0.0007987951,0.24363953,0.018539276,0.000024743154],"about_ca_topic_score_codex":0.0098955175,"about_ca_topic_score_gemma":0.01329402,"teacher_disagreement_score":0.0098955175,"about_ca_system_score_codex":0.0019658005,"about_ca_system_score_gemma":0.002833168,"threshold_uncertainty_score":0.029627383},"labels":[],"label_agreement":null},{"id":"W4400079181","doi":"10.1109/tse.2024.3419919","title":"A Scalable t-Wise Coverage Estimator: Algorithms and Applications","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Statistical Methods and Inference","field":"Mathematics","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Algorithm; Scalability; Estimator; Database; Mathematics; Statistics","score_opus":0.028333731620727668,"score_gpt":0.3048803617272769,"score_spread":0.27654663010654923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400079181","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030216284,0.00042756012,0.9935021,0.00024827305,0.000036398837,0.000090080066,0.00019818117,0.0014694837,0.0010062939],"genre_scores_gemma":[0.12219469,0.00081833586,0.8712774,0.00033257392,0.000200199,0.00066999363,0.0014627846,0.0005338352,0.0025101495],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982161,0.0004157956,0.000121759665,0.000516056,0.0005577364,0.00017253711],"domain_scores_gemma":[0.99351776,0.0039868555,0.0006195572,0.0006944968,0.00093953585,0.0002417668],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019149829,0.0017904565,0.001634865,0.001977471,0.00060232135,0.0018076168,0.0022972089,0.0014846432,0.005665289],"category_scores_gemma":[0.017218191,0.00066741084,0.001307784,0.0021001217,0.00084564666,0.002350761,0.0028282488,0.0025134783,0.002210076],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032453847,0.00015563908,0.0033121079,0.0004413769,0.00012204886,0.0002640708,0.00016483694,0.5349228,0.0122317085,0.03279428,0.01522676,0.40003985],"study_design_scores_gemma":[0.00001914434,0.00002904476,0.00020460338,0.000025181613,0.000011368149,0.000083390165,0.00001606878,0.9857056,0.0013744696,0.011068484,0.0014502715,0.000012243967],"about_ca_topic_score_codex":0.0062892884,"about_ca_topic_score_gemma":0.0052076187,"teacher_disagreement_score":0.0062892884,"about_ca_system_score_codex":0.0014421867,"about_ca_system_score_gemma":0.0028953198,"threshold_uncertainty_score":0.01895231},"labels":[],"label_agreement":null},{"id":"W4400276346","doi":"10.1109/tse.2024.3422369","title":"Characterizing the Prevalence, Distribution, and Duration of Stale Reviewer Recommendations","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Species Distribution and Climate Change","field":"Environmental Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Waterloo","funders":"","keywords":"Computer science; Duration (music)","score_opus":0.014897566962101812,"score_gpt":0.23222201763170797,"score_spread":0.21732445066960615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400276346","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.979788,0.0014122385,0.015244208,0.00069681776,0.000063417865,0.00012515481,0.00054733583,0.00057705556,0.0015457512],"genre_scores_gemma":[0.98783255,0.00034372628,0.009662426,0.00011397999,0.000053865602,0.00008219008,0.00083353923,0.00009562838,0.0009821287],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9777893,0.007580127,0.0024502377,0.005320717,0.0054726317,0.0013868825],"domain_scores_gemma":[0.5226691,0.34170115,0.06862643,0.026155703,0.03410807,0.0067395996],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.022759693,0.0006604248,0.0008172947,0.00513407,0.0016115614,0.0027014036,0.0017183536,0.0020799846,0.0009452412],"category_scores_gemma":[0.2105638,0.0006545491,0.000684397,0.00318911,0.0011787815,0.004080955,0.001448832,0.002201268,0.00077421375],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004362504,0.00027628426,0.88725317,0.00044289496,0.0002666669,0.00048988924,0.00383831,0.008367019,0.005336846,0.0009220303,0.0031441979,0.08922645],"study_design_scores_gemma":[0.00010462162,0.0011131298,0.7971615,0.0003285568,0.00042939474,0.003023523,0.0062641506,0.16845767,0.011019886,0.0032692293,0.008566521,0.00026188517],"about_ca_topic_score_codex":0.007318388,"about_ca_topic_score_gemma":0.016069628,"teacher_disagreement_score":0.9772403,"about_ca_system_score_codex":0.0013985144,"about_ca_system_score_gemma":0.0023566228,"threshold_uncertainty_score":0.120366216},"labels":[],"label_agreement":null},{"id":"W4400351643","doi":"10.1109/tse.2024.3423712","title":"Revisiting the Performance of Deep Learning-Based Vulnerability Detection on Realistic Datasets","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Network Security and Intrusion Detection","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Vulnerability (computing); Artificial intelligence; Deep learning; Data science; Machine learning; Computer security","score_opus":0.009199555430048047,"score_gpt":0.2236496439675864,"score_spread":0.21445008853753836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400351643","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8545731,0.012622786,0.08196401,0.0051189633,0.0010319867,0.00040075282,0.01848553,0.01566294,0.010139956],"genre_scores_gemma":[0.90449476,0.0013045482,0.05080328,0.0009065322,0.00011381099,0.00017477755,0.038788028,0.0003191917,0.0030949796],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.99629706,0.0012129039,0.00034461042,0.001066993,0.00071749726,0.0003608546],"domain_scores_gemma":[0.99150085,0.005117729,0.00049109856,0.0014027879,0.0012291442,0.0002583283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074284323,0.0030943851,0.0012359875,0.002778263,0.00081563176,0.00231976,0.0026582212,0.0025364633,0.0012410418],"category_scores_gemma":[0.023506114,0.00059455587,0.0012584542,0.0018668895,0.0010528408,0.0037143007,0.0022497152,0.0031259286,0.0012758978],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000985712,0.0010467643,0.0519693,0.00103495,0.00096179225,0.00034158304,0.00028776468,0.60521764,0.0068247807,0.0023316154,0.044204593,0.28479356],"study_design_scores_gemma":[0.000042550855,0.00028315303,0.00558579,0.00011844677,0.00006115694,0.00012887805,0.00012481275,0.9796125,0.0072330767,0.0025300218,0.00423648,0.000043199794],"about_ca_topic_score_codex":0.030397505,"about_ca_topic_score_gemma":0.027338343,"teacher_disagreement_score":0.030397505,"about_ca_system_score_codex":0.0025618845,"about_ca_system_score_gemma":0.001711248,"threshold_uncertainty_score":0.060441136},"labels":[],"label_agreement":null},{"id":"W4400647264","doi":"10.1109/tse.2024.3428324","title":"Towards Efficient Fine-Tuning of Language Models With Organizational Data for Automated Software Review","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Software engineering; Software; Programming language; Data science","score_opus":0.024687243253182226,"score_gpt":0.27963316747094596,"score_spread":0.25494592421776374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400647264","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09542399,0.0037277166,0.86859936,0.001632009,0.000241258,0.00057171885,0.0010892415,0.026985167,0.0017296075],"genre_scores_gemma":[0.5518605,0.0007948928,0.4355762,0.0013502813,0.00019006156,0.0007630682,0.0052900068,0.00094037026,0.003234671],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9945503,0.002802685,0.0003320625,0.0013313191,0.00074815744,0.0002355507],"domain_scores_gemma":[0.98044306,0.012307264,0.0015245636,0.0020939456,0.0028496187,0.00078160304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059137875,0.0016712563,0.0016724875,0.0026833836,0.0006740881,0.0018034085,0.0035014765,0.0021220036,0.0015303778],"category_scores_gemma":[0.032909315,0.00089931017,0.0014978289,0.001310255,0.0008327045,0.004214533,0.002694579,0.0036359848,0.001743206],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000580661,0.0009776854,0.012274203,0.0010960257,0.00033175544,0.00033283286,0.0010044737,0.21396476,0.023921054,0.0030923332,0.02247512,0.719949],"study_design_scores_gemma":[0.000044353383,0.000118338416,0.000690258,0.000033228305,0.000030191855,0.000062409345,0.00008285354,0.98906404,0.0046251835,0.0030903295,0.0021356046,0.000023206057],"about_ca_topic_score_codex":0.009124827,"about_ca_topic_score_gemma":0.018876992,"teacher_disagreement_score":0.009124827,"about_ca_system_score_codex":0.0017189933,"about_ca_system_score_gemma":0.004149133,"threshold_uncertainty_score":0.03127551},"labels":[],"label_agreement":null},{"id":"W4400978763","doi":"10.1109/tse.2024.3433463","title":"Assessing Evaluation Metrics for Neural Test Oracle Generation","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Computer science; Oracle; Test (biology); Artificial neural network; Machine learning; Artificial intelligence; Data mining; Software engineering","score_opus":0.0715888736640746,"score_gpt":0.32531120179375506,"score_spread":0.25372232812968043,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400978763","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7561023,0.0070629446,0.21385626,0.0009375013,0.0002648613,0.0004984291,0.0020189257,0.012121408,0.0071372744],"genre_scores_gemma":[0.9271518,0.00041105808,0.06611537,0.00019817583,0.000051913004,0.00022887294,0.004573859,0.00031137688,0.0009575615],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9812439,0.00940529,0.0019100762,0.0022421458,0.0046207807,0.00057781604],"domain_scores_gemma":[0.9242682,0.05349312,0.0053389757,0.0059401887,0.009645438,0.0013141866],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0163487,0.0020596148,0.0011184674,0.00547723,0.00042894727,0.0017256219,0.0026717172,0.001831361,0.0010926625],"category_scores_gemma":[0.07321007,0.00048502474,0.00088132307,0.002708521,0.0009812658,0.0031634367,0.0015279104,0.0015608416,0.0004443592],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001117059,0.0006978048,0.05391304,0.00062249345,0.0007740631,0.0001518261,0.00019860655,0.58638835,0.00411688,0.002536953,0.006851782,0.34263113],"study_design_scores_gemma":[0.00004626533,0.0005470799,0.0045945193,0.0000372788,0.000055152526,0.0000606277,0.00004131041,0.98827296,0.004734738,0.0010869529,0.0004993793,0.000023735629],"about_ca_topic_score_codex":0.008149184,"about_ca_topic_score_gemma":0.009440368,"teacher_disagreement_score":0.0163487,"about_ca_system_score_codex":0.0040125526,"about_ca_system_score_gemma":0.0015552833,"threshold_uncertainty_score":0.086461246},"labels":[],"label_agreement":null},{"id":"W4401073387","doi":"10.1109/tse.2024.3435067","title":"Mitigating the Uncertainty and Imprecision of Log-Based Code Coverage Without Requiring Additional Logging Statements","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Waterloo","funders":"","keywords":"Computer science; Logging; Code (set theory); Programming language; Data mining; Set (abstract data type)","score_opus":0.019196045384707247,"score_gpt":0.28167165167033514,"score_spread":0.2624756062856279,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401073387","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08253413,0.0006094046,0.9052518,0.0010726039,0.000061547624,0.00013611888,0.000599075,0.0067502274,0.0029850993],"genre_scores_gemma":[0.78267586,0.0003144323,0.21299684,0.0005004003,0.00009591638,0.0002378204,0.0011374921,0.0010523032,0.0009889533],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98171633,0.0044151624,0.0010729171,0.0028045133,0.009232014,0.00075904164],"domain_scores_gemma":[0.8632076,0.0845481,0.013448546,0.026939152,0.011051242,0.0008054895],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0095301615,0.0015122981,0.0013071501,0.0038493648,0.00082285,0.0034256966,0.0030548626,0.001357232,0.0009892347],"category_scores_gemma":[0.10093305,0.0012576196,0.0008183238,0.002349546,0.0021114082,0.007089182,0.00443108,0.0030085538,0.0004791465],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013267428,0.00046449498,0.07460446,0.0008760934,0.0004260296,0.00072428514,0.0023239972,0.30468017,0.0371257,0.03763637,0.007647926,0.5321637],"study_design_scores_gemma":[0.00005988428,0.0002501719,0.014634327,0.00023398381,0.00011861003,0.0004969799,0.00035153748,0.875133,0.046600565,0.0530657,0.008902496,0.00015284844],"about_ca_topic_score_codex":0.0045655393,"about_ca_topic_score_gemma":0.0054664663,"teacher_disagreement_score":0.0095301615,"about_ca_system_score_codex":0.0015105257,"about_ca_system_score_gemma":0.0025112163,"threshold_uncertainty_score":0.050400913},"labels":[],"label_agreement":null},{"id":"W4401211363","doi":"10.1109/tse.2024.3436623","title":"Long Live the Image: On Enabling Resilient Production Database Containers for Microservice Applications","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Queen's University; Queen's University Belfast","keywords":"Computer science; Production (economics); Database; Microservices; Software engineering; World Wide Web; Operating system; Cloud computing","score_opus":0.010017254744161024,"score_gpt":0.24516179219632728,"score_spread":0.23514453745216626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401211363","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11107084,0.0022223662,0.84449196,0.0029841403,0.00044035426,0.0004029081,0.00010984626,0.016025318,0.022252236],"genre_scores_gemma":[0.68233615,0.001959479,0.29516724,0.0017774059,0.00016974394,0.0002606719,0.00050685083,0.0016555057,0.016166985],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989826,0.00014041633,0.000061548104,0.00018622876,0.00035445252,0.00027469386],"domain_scores_gemma":[0.99856865,0.00020578435,0.00012286607,0.0005155588,0.0003982019,0.00018897056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011128479,0.00056549074,0.00035859214,0.0004528627,0.00093620794,0.0023982762,0.00273125,0.0010691248,0.002648199],"category_scores_gemma":[0.0027702844,0.00041293606,0.00059094833,0.0004919443,0.0013621314,0.007052517,0.005178364,0.0020433054,0.00072218815],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009699367,0.0006658897,0.009085196,0.0012246133,0.00012435084,0.0022680522,0.00547903,0.03209792,0.13871983,0.2712019,0.04022602,0.49793717],"study_design_scores_gemma":[0.00013492478,0.001489519,0.0050583864,0.00049256644,0.00024736315,0.0023720143,0.002697824,0.39893168,0.24340014,0.05859159,0.28631875,0.0002652715],"about_ca_topic_score_codex":0.0027376893,"about_ca_topic_score_gemma":0.0019474076,"teacher_disagreement_score":0.0027376893,"about_ca_system_score_codex":0.0009052882,"about_ca_system_score_gemma":0.0011894357,"threshold_uncertainty_score":0.008859098},"labels":[],"label_agreement":null},{"id":"W4401328698","doi":"10.1109/tse.2024.3438119","title":"AddressWatcher: Sanitizer-Based Localization of Memory Leak Fixes","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Data Storage Technologies","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Hand sanitizer; Leak; Memory leak; Embedded system; Leak detection; Programming language; Memory management; Overlay; Engineering","score_opus":0.0139245237817579,"score_gpt":0.23100124283280993,"score_spread":0.21707671905105203,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401328698","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16152875,0.0017900758,0.42681232,0.0004931197,0.00029888775,0.00082590204,0.0011635816,0.39851803,0.00856932],"genre_scores_gemma":[0.63149804,0.0005276682,0.34684864,0.000593455,0.000078144774,0.0005565023,0.0026303772,0.009037516,0.008229742],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99678636,0.00057324854,0.00028739998,0.0007305351,0.0012717098,0.0003508063],"domain_scores_gemma":[0.9927532,0.0022614074,0.0010587709,0.0024011626,0.0012640163,0.0002614234],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002431305,0.0021932675,0.0010584595,0.0018723559,0.00074342295,0.0013152149,0.0050495113,0.0016627937,0.005919467],"category_scores_gemma":[0.010776324,0.0011602908,0.00086413254,0.00085460284,0.0013510083,0.0042060786,0.003157183,0.0018909059,0.0019365085],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002971116,0.0013667172,0.025228297,0.0016427592,0.00041447338,0.0007692764,0.0011697707,0.059483215,0.1306589,0.011027913,0.045662556,0.71960497],"study_design_scores_gemma":[0.00055228814,0.0023176728,0.009113464,0.00016087745,0.00040376338,0.00072334154,0.0002909634,0.6428662,0.29244557,0.0067521306,0.044115618,0.00025819667],"about_ca_topic_score_codex":0.0043866164,"about_ca_topic_score_gemma":0.0047470666,"teacher_disagreement_score":0.005919467,"about_ca_system_score_codex":0.0010035469,"about_ca_system_score_gemma":0.0024508995,"threshold_uncertainty_score":0.01980257},"labels":[],"label_agreement":null},{"id":"W4401537664","doi":"10.1109/tse.2024.3443741","title":"Predicting the First Response Latency of Maintainers and Contributors in Pull Requests","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Concordia University","funders":"","keywords":"Computer science; Latency (audio); Parallel computing; Telecommunications","score_opus":0.005120476108220783,"score_gpt":0.2077019446545747,"score_spread":0.20258146854635392,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401537664","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9692542,0.00046128273,0.025157038,0.00036441794,0.00012147781,0.00011864148,0.0011983337,0.0021682908,0.0011563742],"genre_scores_gemma":[0.98291713,0.0001335089,0.012604425,0.000060917075,0.00008415324,0.00007652082,0.0025746126,0.00009981823,0.0014490113],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99725527,0.00055729947,0.00025567302,0.0006243696,0.00089468807,0.00041263865],"domain_scores_gemma":[0.97030556,0.014892176,0.005179998,0.0016462888,0.0058297385,0.0021461744],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004971646,0.0012390917,0.00077661296,0.0022299807,0.00047292426,0.0014690575,0.00073753716,0.001005811,0.00074190204],"category_scores_gemma":[0.027116675,0.0003406709,0.00057454186,0.0013492326,0.00026632877,0.0017391564,0.00072068797,0.0014591792,0.001312727],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015784135,0.0008654088,0.68441826,0.00046891405,0.00021051326,0.000621409,0.00091106445,0.079121865,0.023355447,0.00066799164,0.012141759,0.195639],"study_design_scores_gemma":[0.000035818965,0.0005505929,0.162841,0.000043610613,0.00007718847,0.00038260056,0.000494942,0.8199873,0.011743024,0.00085447996,0.0029274065,0.00006206552],"about_ca_topic_score_codex":0.0056904457,"about_ca_topic_score_gemma":0.006523043,"teacher_disagreement_score":0.0056904457,"about_ca_system_score_codex":0.00071152154,"about_ca_system_score_gemma":0.0015016476,"threshold_uncertainty_score":0.02629286},"labels":[],"label_agreement":null},{"id":"W4402040489","doi":"10.1109/tse.2024.3452595","title":"RLocator: Reinforcement Learning for Bug Localization","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; University of Waterloo","funders":"","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Software bug; Programming language; Software engineering; Human–computer interaction; Machine learning; Software","score_opus":0.014284326079011332,"score_gpt":0.2518457570048798,"score_spread":0.23756143092586848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402040489","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053061582,0.0009474588,0.93197376,0.000623418,0.00010265249,0.00022154402,0.00027082293,0.010954749,0.0018440263],"genre_scores_gemma":[0.7527251,0.00028707559,0.2426705,0.00061754766,0.00009090747,0.0004996429,0.0007475874,0.00047404366,0.001887609],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977804,0.0010489271,0.00010998561,0.00048536394,0.00041559903,0.00015975471],"domain_scores_gemma":[0.989824,0.007621112,0.0006866252,0.0006490924,0.0009081605,0.00031098598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042398334,0.0016554625,0.0015700586,0.0012356055,0.0003895669,0.0008230002,0.0029956745,0.0015801821,0.0018295777],"category_scores_gemma":[0.017888134,0.0007096471,0.0008031076,0.0007386355,0.0010693438,0.0015995502,0.0016213314,0.0023970865,0.0006159026],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000249632,0.00036619502,0.0052294927,0.00018987177,0.00012865588,0.00010505915,0.000097735756,0.74808663,0.0024082947,0.003075335,0.0065800766,0.233483],"study_design_scores_gemma":[0.00002495476,0.000048813494,0.00015247449,0.0000061619917,0.000008180927,0.000012171005,0.00000422116,0.99762434,0.0005006438,0.0013969865,0.00021561056,0.0000054791885],"about_ca_topic_score_codex":0.0051592193,"about_ca_topic_score_gemma":0.006033282,"teacher_disagreement_score":0.0051592193,"about_ca_system_score_codex":0.001680993,"about_ca_system_score_gemma":0.002093189,"threshold_uncertainty_score":0.022422612},"labels":[],"label_agreement":null},{"id":"W4402661245","doi":"10.1109/tse.2024.3461657","title":"D<sup>3</sup>: Differential Testing of Distributed Deep Learning With Model Generation","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Division of Computing and Communication Foundations","keywords":"Computer science; Artificial intelligence; Programming language","score_opus":0.014824459603576702,"score_gpt":0.21919131723217236,"score_spread":0.20436685762859566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402661245","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02394977,0.00008880115,0.9395513,0.00091700297,0.0003235708,0.00022243023,0.0009126747,0.014662194,0.019372173],"genre_scores_gemma":[0.5317449,0.00011273313,0.44384646,0.0019818735,0.00016200876,0.000534597,0.0048709773,0.0029914156,0.013754971],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953928,0.0010756829,0.0005095119,0.00090124767,0.0017312712,0.00038945614],"domain_scores_gemma":[0.9892,0.004475313,0.00047312133,0.0037197412,0.0018700964,0.0002617463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036788357,0.0009066171,0.0005298175,0.00082955364,0.0005591001,0.0017418566,0.0042445627,0.0014722877,0.017934714],"category_scores_gemma":[0.017039806,0.0004286156,0.0011406504,0.0005652162,0.0019238702,0.0023487369,0.002263523,0.0021184788,0.0039833444],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016979149,0.00068161933,0.01321409,0.0004889234,0.00014820659,0.0014335314,0.00032253872,0.10117956,0.059908003,0.1382799,0.096103445,0.58654237],"study_design_scores_gemma":[0.0002020449,0.00038308345,0.00217632,0.000055701355,0.00004006967,0.00066698645,0.00007909438,0.7584573,0.13577542,0.068627514,0.033475265,0.00006122161],"about_ca_topic_score_codex":0.0031847185,"about_ca_topic_score_gemma":0.0042319633,"teacher_disagreement_score":0.017934714,"about_ca_system_score_codex":0.0014318492,"about_ca_system_score_gemma":0.0017295742,"threshold_uncertainty_score":0.05999756},"labels":[],"label_agreement":null},{"id":"W4402979393","doi":"10.1109/tse.2024.3469582","title":"LTM: Scalable and Black-Box Similarity-Based Test Suite Minimization Based on Language Models","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trent University; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada; Science Foundation Ireland","keywords":"Computer science; Test suite; Suite; Scalability; Black box; Minification; Similarity (geometry); Test (biology); Software testing; Programming language; Artificial intelligence; Natural language processing; Test case; Software; Machine learning; Operating system","score_opus":0.012067979201598519,"score_gpt":0.22770173116824316,"score_spread":0.21563375196664464,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402979393","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031785652,0.0003180002,0.9539198,0.00018508993,0.00003783467,0.00022951531,0.0003411955,0.012057973,0.001124961],"genre_scores_gemma":[0.39652836,0.00017620507,0.5952758,0.00037782153,0.00005999994,0.0006499056,0.003016325,0.0013652277,0.002550332],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966059,0.0008353331,0.00023753673,0.00066258304,0.0013909796,0.00026775317],"domain_scores_gemma":[0.99552,0.0019641798,0.00065144454,0.00083419535,0.0008231046,0.00020703071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016065646,0.001679135,0.0014687987,0.0019453132,0.0005230513,0.0012289311,0.0030413985,0.0011825558,0.0022722355],"category_scores_gemma":[0.010397906,0.00057719706,0.0016731913,0.0011800139,0.0009957781,0.0023583544,0.0026070967,0.0017391035,0.0010189726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052190194,0.0004163557,0.004876078,0.0004906138,0.00022026907,0.00041257232,0.00023204515,0.3393781,0.032141805,0.0124070505,0.008228276,0.6006749],"study_design_scores_gemma":[0.00003225071,0.00016019013,0.0003633985,0.000012245695,0.000021235279,0.00010990687,0.000030004861,0.9821755,0.009828903,0.0062555247,0.0009962813,0.000014587585],"about_ca_topic_score_codex":0.0050331466,"about_ca_topic_score_gemma":0.006445027,"teacher_disagreement_score":0.0050331466,"about_ca_system_score_codex":0.0015425693,"about_ca_system_score_gemma":0.0028987718,"threshold_uncertainty_score":0.011192143},"labels":[],"label_agreement":null},{"id":"W4403060990","doi":"10.1109/tse.2024.3472476","title":"FlakyFix: Using Large Language Models for Predicting Flaky Test Fix Categories and Test Code Repair","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Ottawa","funders":"Science Foundation Ireland","keywords":"Computer science; Test (biology); Programming language; Code (set theory); Code coverage; Software engineering; Reliability engineering; Software; Engineering","score_opus":0.017848771159971646,"score_gpt":0.2560316308607052,"score_spread":0.23818285970073355,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403060990","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40254417,0.0022846542,0.48961008,0.0017724764,0.00037093167,0.00056620984,0.013530003,0.08664207,0.0026794255],"genre_scores_gemma":[0.7101169,0.00030427313,0.25890473,0.00059677137,0.00007660829,0.00052267156,0.025082335,0.001397629,0.0029979805],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986584,0.00039223622,0.00008742434,0.000527372,0.00022336705,0.000111216024],"domain_scores_gemma":[0.992694,0.004812276,0.000622818,0.00078034523,0.00077725394,0.00031326726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002000266,0.0022184532,0.0008043775,0.0023329507,0.000510273,0.0011345375,0.0019239719,0.0019511944,0.0019694094],"category_scores_gemma":[0.010637897,0.00055427593,0.0014841435,0.00088575174,0.00062636845,0.0021545608,0.0013298234,0.0031485015,0.0015094983],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012101313,0.0012451935,0.06855758,0.00087334437,0.0004568301,0.0010673669,0.0007733957,0.37742865,0.019940326,0.002850067,0.037881806,0.48771533],"study_design_scores_gemma":[0.000044093453,0.00015821887,0.003011846,0.000032931664,0.000040091567,0.00009059631,0.00006794455,0.9862278,0.0054253256,0.0027822193,0.002078792,0.00004023271],"about_ca_topic_score_codex":0.016934121,"about_ca_topic_score_gemma":0.030209368,"teacher_disagreement_score":0.016934121,"about_ca_system_score_codex":0.0013902454,"about_ca_system_score_gemma":0.0017391375,"threshold_uncertainty_score":0.03367114},"labels":[],"label_agreement":null},{"id":"W4403210636","doi":"10.1109/tse.2024.3475375","title":"Exploring the Effectiveness of LLMs in Automated Logging Statement Generation: An Empirical Study","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Basic and Applied Basic Research Foundation of Guangdong Province; Science Foundation Ireland","keywords":"Computer science; Logging; Statement (logic); Empirical research; Data science; Software engineering; Computer security; Risk analysis (engineering); Business; Forestry; Statistics; Political science","score_opus":0.07912131087874133,"score_gpt":0.3136969886310401,"score_spread":0.2345756777522988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403210636","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98036045,0.0012700368,0.00503933,0.0007787839,0.000041838954,0.0004294802,0.0055494597,0.0034958657,0.0030347113],"genre_scores_gemma":[0.9391948,0.000539251,0.03411618,0.00038343464,0.000037544734,0.00047429683,0.023581017,0.0005031043,0.0011704847],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98821825,0.006539221,0.0009905623,0.0018610811,0.0020934658,0.00029732403],"domain_scores_gemma":[0.7797922,0.18560392,0.010135733,0.013274808,0.008808848,0.0023845688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015387476,0.0008260766,0.0005586838,0.0023370853,0.00088689185,0.0023839762,0.0021273457,0.0015383104,0.001520216],"category_scores_gemma":[0.11772216,0.0005278798,0.0007531465,0.0018763347,0.001103898,0.004699909,0.0014343652,0.0024311533,0.0013097023],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00509445,0.010206451,0.4711172,0.004957317,0.00058734306,0.00076516723,0.009014988,0.023804773,0.011512727,0.0023338941,0.051176056,0.40942964],"study_design_scores_gemma":[0.0019418319,0.0075665363,0.40298235,0.0014578394,0.0010578561,0.0017818308,0.011640527,0.4447099,0.032160994,0.0068596816,0.0874034,0.00043723045],"about_ca_topic_score_codex":0.005033328,"about_ca_topic_score_gemma":0.0067816563,"teacher_disagreement_score":0.015387476,"about_ca_system_score_codex":0.001094079,"about_ca_system_score_gemma":0.0016112346,"threshold_uncertainty_score":0.081377745},"labels":[],"label_agreement":null},{"id":"W4403511313","doi":"10.1109/tse.2024.3482984","title":"TEASMA: A Practical Methodology for Test Adequacy Assessment of Deep Neural Networks","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Ottawa","funders":"Huawei Technologies; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Science Foundation Ireland","keywords":"Computer science; Artificial neural network; Test (biology); Artificial intelligence; Machine learning; Reliability engineering; Software engineering; Engineering","score_opus":0.03694737320835111,"score_gpt":0.34241031545145006,"score_spread":0.30546294224309894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403511313","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034253918,0.00026130807,0.9498095,0.00014413746,0.000052318464,0.0003962715,0.0009006629,0.012421541,0.001760362],"genre_scores_gemma":[0.26638913,0.00012621624,0.7287574,0.00015131786,0.000032014304,0.0011630014,0.0016994524,0.0007950743,0.00088642054],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9927362,0.002350373,0.0009995585,0.0008892016,0.0027490782,0.00027561988],"domain_scores_gemma":[0.970574,0.017654607,0.0036254001,0.0029138632,0.004874863,0.0003572478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009175902,0.0020850424,0.0008224024,0.0055722627,0.00049713603,0.0014977589,0.0021512224,0.0013396397,0.0037371684],"category_scores_gemma":[0.054914247,0.0006274105,0.0011741683,0.0014733135,0.0010957145,0.0018097708,0.0023870992,0.0016222084,0.0006819736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063375104,0.00032113248,0.028460763,0.00095815735,0.00044611457,0.00054113794,0.0006365847,0.37902874,0.024600502,0.020620547,0.011278543,0.532474],"study_design_scores_gemma":[0.000057486843,0.00028903375,0.0034349903,0.00011030407,0.00004040356,0.00022248465,0.0001051998,0.9627237,0.018429866,0.011138729,0.0033953474,0.000052519474],"about_ca_topic_score_codex":0.0034287826,"about_ca_topic_score_gemma":0.0043553296,"teacher_disagreement_score":0.009175902,"about_ca_system_score_codex":0.001122662,"about_ca_system_score_gemma":0.0021336745,"threshold_uncertainty_score":0.04852736},"labels":[],"label_agreement":null},{"id":"W4403635895","doi":"10.1109/tse.2024.3484586","title":"Refactoring-Aware Block Tracking in Commit History","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code refactoring; Computer science; Commit; Block (permutation group theory); Programming language; Software engineering; Database; Software","score_opus":0.01952255369704392,"score_gpt":0.23632262482171681,"score_spread":0.2168000711246729,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403635895","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20872681,0.004636234,0.53194755,0.00067003095,0.00035252364,0.0005572064,0.01546168,0.23207062,0.005577373],"genre_scores_gemma":[0.5561901,0.0011065201,0.39164034,0.00031622322,0.00011536499,0.00034231527,0.03949745,0.006205742,0.004585888],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9964485,0.0005239846,0.00040020648,0.0010786146,0.0013628597,0.00018581057],"domain_scores_gemma":[0.9809871,0.006446318,0.0028352195,0.005057792,0.0042660553,0.00040743843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002606224,0.001185277,0.0008518598,0.005468538,0.00045700764,0.001804505,0.0013542862,0.00091253896,0.0012774057],"category_scores_gemma":[0.02717429,0.00064801,0.00064564095,0.0028660176,0.00040210233,0.0031124197,0.0015965686,0.0010625679,0.0018153285],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055393076,0.000213141,0.100355454,0.0013889411,0.00017678471,0.0005748653,0.0015222853,0.019327296,0.030917685,0.0034634597,0.042727903,0.79877836],"study_design_scores_gemma":[0.00017408645,0.0006171641,0.082888514,0.00055505143,0.0003086576,0.0018565671,0.0006498584,0.66435117,0.13895647,0.014057836,0.09528254,0.00030212261],"about_ca_topic_score_codex":0.00678495,"about_ca_topic_score_gemma":0.010923259,"teacher_disagreement_score":0.00678495,"about_ca_system_score_codex":0.00046382204,"about_ca_system_score_gemma":0.001670305,"threshold_uncertainty_score":0.0137832165},"labels":[],"label_agreement":null},{"id":"W4403918536","doi":"10.1109/tse.2024.3488525","title":"AIM: Automated Input Set Minimization for Metamorphic Security Testing","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"H2020 European Research Council; Natural Sciences and Engineering Research Council of Canada; Horizon 2020 Framework Programme; Science Foundation Ireland; Université du Luxembourg; University of Ottawa","keywords":"Computer science; Set (abstract data type); Minification; Programming language; Theoretical computer science; Software engineering; Data mining; Algorithm","score_opus":0.02010568178901174,"score_gpt":0.25042375612891415,"score_spread":0.23031807433990242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403918536","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10049085,0.00041025257,0.87953436,0.00037927506,0.00004577678,0.00034855853,0.0004491052,0.012319156,0.006022773],"genre_scores_gemma":[0.41326123,0.00012382571,0.5817349,0.00025813084,0.000020577008,0.00037115155,0.0009698604,0.00073345384,0.0025268984],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99866307,0.0004803677,0.00006193604,0.00021492409,0.0004269371,0.00015279447],"domain_scores_gemma":[0.9980574,0.0012376558,0.00014071655,0.00026067544,0.00023263534,0.00007081592],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011188821,0.0013466098,0.0007480847,0.0011697738,0.0004208686,0.00072026264,0.001996461,0.0011328345,0.0042633694],"category_scores_gemma":[0.0040273257,0.00042491456,0.00095745584,0.0006493918,0.0007061707,0.0010122485,0.0015036309,0.0013943061,0.00076750416],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005159838,0.0005854532,0.005328612,0.0005814733,0.00013340662,0.000302548,0.0002003427,0.5002616,0.047583517,0.010586406,0.008553738,0.42536682],"study_design_scores_gemma":[0.000058115944,0.00015086253,0.00067133986,0.000024813213,0.000027156744,0.00010326571,0.000036038546,0.9789104,0.013616478,0.0046023726,0.0017877667,0.000011391236],"about_ca_topic_score_codex":0.001959509,"about_ca_topic_score_gemma":0.003219257,"teacher_disagreement_score":0.0042633694,"about_ca_system_score_codex":0.0008489107,"about_ca_system_score_gemma":0.0015365507,"threshold_uncertainty_score":0.014262378},"labels":[],"label_agreement":null},{"id":"W4404102001","doi":"10.1109/tse.2024.3491496","title":"SMARLA: A Safety Monitoring Approach for Deep Reinforcement Learning Agents","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Science Foundation Ireland","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Machine learning; Software engineering","score_opus":0.01605624727510781,"score_gpt":0.25027956476942625,"score_spread":0.23422331749431843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404102001","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00785913,0.00013183166,0.98212034,0.00028164696,0.00005892673,0.00010568331,0.00011048393,0.0071728737,0.002159027],"genre_scores_gemma":[0.4665076,0.00014883274,0.52528286,0.0004827265,0.000068374145,0.00048686532,0.0003825749,0.00051668327,0.006123519],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99890375,0.0003363695,0.00007464801,0.00022677198,0.0003350437,0.00012334106],"domain_scores_gemma":[0.99845695,0.00061680237,0.00025200142,0.00020551147,0.00033824192,0.00013051486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021526918,0.001146469,0.00097631646,0.0006847046,0.00055587845,0.0010275244,0.0028633974,0.0014208434,0.0039326735],"category_scores_gemma":[0.004901146,0.00069090846,0.0008282402,0.00024543126,0.0008446661,0.0014666106,0.0020221246,0.0024529148,0.0009440963],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031941262,0.00020549461,0.0016192192,0.0001333757,0.00010923223,0.00014118322,0.00014558375,0.7462417,0.004991782,0.016386334,0.0073308237,0.22237585],"study_design_scores_gemma":[0.000018648174,0.000030315454,0.00005501049,0.000006475904,0.000006211652,0.000008911282,0.0000035285482,0.99413943,0.0011300453,0.0034822437,0.0011140098,0.0000052159403],"about_ca_topic_score_codex":0.005575088,"about_ca_topic_score_gemma":0.006611646,"teacher_disagreement_score":0.005575088,"about_ca_system_score_codex":0.0014211269,"about_ca_system_score_gemma":0.0023315505,"threshold_uncertainty_score":0.013156116},"labels":[],"label_agreement":null},{"id":"W4404317069","doi":"10.1109/tse.2024.3470368","title":"Scoping Software Engineering for AI: The TSE Perspective","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; University of Alberta; Queen's University; University of Toronto","funders":"","keywords":"Computer science; Software engineering; Perspective (graphical); Software development; Software; Programming language; Artificial intelligence","score_opus":0.013694603385929241,"score_gpt":0.25763863923195046,"score_spread":0.24394403584602123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404317069","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028931568,0.12031658,0.119155094,0.55472887,0.092927665,0.00025264345,0.00015201473,0.00032235906,0.109251596],"genre_scores_gemma":[0.21388745,0.23615925,0.109763086,0.13464603,0.19526556,0.0011674277,0.0005350393,0.0011861911,0.107389964],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9670784,0.02037909,0.0035797327,0.001912385,0.005910221,0.0011402515],"domain_scores_gemma":[0.8202537,0.13368438,0.007833165,0.009474794,0.02342381,0.005330106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.055379555,0.000877978,0.0011264607,0.008254416,0.0040809126,0.01923783,0.0024299994,0.009804505,0.0069432626],"category_scores_gemma":[0.13607809,0.00065284636,0.0011607402,0.0066265985,0.02087826,0.01681613,0.008259543,0.009401423,0.0025638663],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000023995533,0.000021085401,0.00022177865,0.0015504557,0.00003288197,0.00038081236,0.0041159145,0.00066420774,0.00019873209,0.8297703,0.086215824,0.07680396],"study_design_scores_gemma":[0.000013388757,0.000024343834,0.000116101386,0.003630896,0.000018027471,0.00033043843,0.0021019878,0.000719367,0.00021154978,0.4656753,0.5271363,0.000022237617],"about_ca_topic_score_codex":0.0012569071,"about_ca_topic_score_gemma":0.0018978943,"teacher_disagreement_score":0.055379555,"about_ca_system_score_codex":0.0043051913,"about_ca_system_score_gemma":0.013237346,"threshold_uncertainty_score":0.29287857},"labels":[],"label_agreement":null},{"id":"W4404609282","doi":"10.1109/tse.2024.3504286","title":"On Inter-Dataset Code Duplication and Data Leakage in Large Language Models","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University; McGill University","funders":"","keywords":"Computer science; Programming language; Code (set theory); Software bug; Data mining; Software","score_opus":0.025993329642164737,"score_gpt":0.2756331319817165,"score_spread":0.24963980233955177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404609282","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3296938,0.004565615,0.6286742,0.01354997,0.0005207661,0.00065116375,0.0047483374,0.012854946,0.0047411886],"genre_scores_gemma":[0.833484,0.0008029927,0.14889881,0.0029825754,0.00038314963,0.0006379576,0.006726351,0.0015002065,0.0045840065],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.9706745,0.013999949,0.002330345,0.006844714,0.004346288,0.0018041177],"domain_scores_gemma":[0.76979864,0.16939682,0.009390376,0.043318123,0.006304268,0.0017917984],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.028724778,0.0014985979,0.002957208,0.0030881942,0.0030637549,0.0045601856,0.0057495544,0.003566684,0.002535812],"category_scores_gemma":[0.15623026,0.001882936,0.002995149,0.00584463,0.0040215175,0.013612632,0.008428127,0.0044343504,0.0010557726],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023954837,0.00091094064,0.059938576,0.00080977794,0.0010049612,0.0015263899,0.0019723775,0.4896173,0.0044682967,0.058665648,0.037367295,0.34132293],"study_design_scores_gemma":[0.00010692558,0.00015567629,0.0025127959,0.00008018933,0.00013072017,0.0004471523,0.00026525228,0.91814876,0.003195175,0.0723694,0.002545991,0.000042081992],"about_ca_topic_score_codex":0.011486677,"about_ca_topic_score_gemma":0.014108108,"teacher_disagreement_score":0.9712752,"about_ca_system_score_codex":0.006384215,"about_ca_system_score_gemma":0.006746542,"threshold_uncertainty_score":0.15191293},"labels":[],"label_agreement":null},{"id":"W4404986931","doi":"10.1109/tse.2025.3631361","title":"Portus: Linking Alloy with SMT-based Finite Model Finding","year":2025,"lang":"en","type":"preprint","venue":"IEEE Transactions on Software Engineering","topic":"Aluminum Alloy Microstructure Properties","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada","keywords":"Alloy; Computer science; Business; Materials science; Metallurgy","score_opus":0.014437106030476488,"score_gpt":0.20637909203541166,"score_spread":0.19194198600493517,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404986931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013408128,0.00009048212,0.9800514,0.00013377889,0.00006653606,0.00010366894,0.00026827998,0.013486164,0.004458907],"genre_scores_gemma":[0.023400487,0.0001890437,0.96708083,0.00017403883,0.000041723848,0.00025324285,0.0013308524,0.003998061,0.0035316835],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963702,0.001160067,0.0003278338,0.0005718287,0.001350288,0.00021981318],"domain_scores_gemma":[0.9957984,0.0024113744,0.00018142349,0.0010157117,0.00051530515,0.000077761986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031120363,0.0024165672,0.0013034294,0.0026142548,0.001077323,0.0031156717,0.0049146595,0.0017047946,0.019873584],"category_scores_gemma":[0.011303688,0.0016958533,0.0037391235,0.0014865112,0.0019329829,0.004120171,0.0053198603,0.0031770018,0.0073919464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041482993,0.0004001255,0.0015980212,0.0015326074,0.00031327,0.00069544325,0.00061246,0.17993864,0.013332325,0.35088477,0.031470727,0.41880685],"study_design_scores_gemma":[0.00009174756,0.00008730001,0.00013764483,0.00016031774,0.000091981296,0.00025629552,0.00010647065,0.7450352,0.019182822,0.17089829,0.06389003,0.000061944134],"about_ca_topic_score_codex":0.0041681044,"about_ca_topic_score_gemma":0.0074863643,"teacher_disagreement_score":0.019873584,"about_ca_system_score_codex":0.0018570801,"about_ca_system_score_gemma":0.0025066494,"threshold_uncertainty_score":0.066483796},"labels":[],"label_agreement":null},{"id":"W4405968021","doi":"10.1109/tse.2024.3519464","title":"<i>Look Before You Leap:</i> An Exploratory Study of Uncertainty Analysis for Large Language Models","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; Japan Science and Technology Agency; Japan Society for the Promotion of Science","keywords":"Computer science; Exploratory analysis; Programming language; Exploratory research; Data science; Software engineering","score_opus":0.016702405548157565,"score_gpt":0.25202852066838805,"score_spread":0.2353261151202305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405968021","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16837496,0.0045240126,0.7834773,0.027589554,0.000388861,0.00050498545,0.00110629,0.0017325219,0.012301472],"genre_scores_gemma":[0.74407494,0.0017816476,0.24380971,0.0042411126,0.00041619907,0.00067152,0.0013552102,0.000761191,0.0028884236],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98900235,0.0070697316,0.00033921088,0.0010455747,0.002293453,0.00024971727],"domain_scores_gemma":[0.8517069,0.13025579,0.0034802365,0.0076790503,0.0059913113,0.000886794],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015577175,0.0010389036,0.0007150014,0.0019274419,0.0013155438,0.004187532,0.002055027,0.0021119134,0.003232282],"category_scores_gemma":[0.11879003,0.00060782215,0.0013239123,0.0015278261,0.0037649986,0.011775701,0.0034539783,0.0063892226,0.0007567147],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000904364,0.0005586096,0.059254542,0.0021739288,0.0009347508,0.0016516044,0.021533696,0.1159967,0.019598255,0.25719944,0.044190105,0.47600406],"study_design_scores_gemma":[0.00008329784,0.00068320805,0.02248672,0.0010278174,0.0001764455,0.0012455018,0.0066548786,0.5717868,0.020463673,0.30791423,0.06708784,0.00038953626],"about_ca_topic_score_codex":0.0061615766,"about_ca_topic_score_gemma":0.006248746,"teacher_disagreement_score":0.015577175,"about_ca_system_score_codex":0.0019725198,"about_ca_system_score_gemma":0.0016442703,"threshold_uncertainty_score":0.08238101},"labels":[],"label_agreement":null},{"id":"W4406610975","doi":"10.1109/tse.2025.3531210","title":"Mecha: A Neural-Symbolic Open-Set Homogeneous Decision Fusion Approach for Zero-Day Malware Similarity Detection","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Blackberry (Canada); Defence Research and Development Canada; Queen's University","funders":"","keywords":"Computer science; Malware; Similarity (geometry); Zero (linguistics); Set (abstract data type); Artificial intelligence; Homogeneous; Data mining; Artificial neural network; Open set; Machine learning; Algorithm; Pattern recognition (psychology); Programming language; Computer security; Mathematics","score_opus":0.015176685173244877,"score_gpt":0.26193065863628473,"score_spread":0.24675397346303984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406610975","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030713161,0.00056080555,0.9612245,0.00036655634,0.00011587614,0.00010599828,0.0002494772,0.004601237,0.0020624737],"genre_scores_gemma":[0.7205959,0.00021281268,0.27165887,0.0006300172,0.00012339791,0.00023550307,0.0011761676,0.00022873573,0.0051385947],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991509,0.00016761228,0.000047131678,0.00025063008,0.00026629816,0.000117345364],"domain_scores_gemma":[0.9988526,0.00048085768,0.00011243913,0.00016332314,0.00030134324,0.00008940515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001495416,0.0012442044,0.0014782907,0.0012008282,0.00067824573,0.0011720138,0.0030369442,0.0015863427,0.0032340444],"category_scores_gemma":[0.0036560302,0.00050701085,0.0010781038,0.00083924236,0.00074043754,0.0023097596,0.0028070814,0.0022688038,0.0011083094],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003751047,0.0003030753,0.0023634566,0.00010072134,0.00017014868,0.00016243706,0.00009676702,0.39584437,0.0066210176,0.0072307773,0.0061933286,0.58053887],"study_design_scores_gemma":[0.0000044149547,0.00003199028,0.00009979488,0.0000031930035,0.0000062650265,0.000013777677,0.000005331103,0.9958223,0.0010298667,0.0026742215,0.00030299186,0.0000057936936],"about_ca_topic_score_codex":0.0050614844,"about_ca_topic_score_gemma":0.0061056037,"teacher_disagreement_score":0.0050614844,"about_ca_system_score_codex":0.0013089086,"about_ca_system_score_gemma":0.0016272166,"threshold_uncertainty_score":0.010819018},"labels":[],"label_agreement":null},{"id":"W4406858164","doi":"10.1109/tse.2025.3533972","title":"On the Workflows and Smells of Leaderboard Operations (LBOps): An Exploratory Study of Foundation Model Leaderboards","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Human Resource Development and Performance Evaluation","field":"Psychology","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Software engineering; Workflow; Programming language; Foundation (evidence); World Wide Web; Database","score_opus":0.04758252356410681,"score_gpt":0.30707135548683356,"score_spread":0.25948883192272676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406858164","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9819411,0.00008921952,0.013866881,0.00024718925,0.00000729047,0.00015715032,0.00029854843,0.00013535931,0.0032572013],"genre_scores_gemma":[0.973918,0.00013891516,0.023009762,0.00009332942,0.0000070736314,0.00019580092,0.0008538021,0.00013406463,0.0016491936],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9960944,0.0020264867,0.00022889125,0.0004992817,0.0008079979,0.000342883],"domain_scores_gemma":[0.96801794,0.020680903,0.0044553876,0.002739713,0.0028706307,0.0012355557],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0061226534,0.0004057395,0.00025305405,0.002386744,0.0017376943,0.0029142478,0.00092431775,0.00063716783,0.0013515815],"category_scores_gemma":[0.027267767,0.0003155136,0.00034893642,0.0023971356,0.0020232582,0.0035321368,0.0025407607,0.0010049059,0.0004549042],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045879194,0.00091216445,0.3359072,0.0008796151,0.00006156079,0.0036356081,0.39941844,0.007382766,0.013361159,0.019966692,0.00797281,0.21004312],"study_design_scores_gemma":[0.00007312002,0.00070456305,0.32555905,0.0010581048,0.000067004185,0.0015583613,0.49371228,0.059557006,0.015402309,0.02600201,0.075991146,0.00031510898],"about_ca_topic_score_codex":0.0068010804,"about_ca_topic_score_gemma":0.009972148,"teacher_disagreement_score":0.99387735,"about_ca_system_score_codex":0.0018411637,"about_ca_system_score_gemma":0.0026544037,"threshold_uncertainty_score":0.032380044},"labels":[],"label_agreement":null},{"id":"W4406858280","doi":"10.1109/tse.2025.3534027","title":"Recovering Traceability Links Between Code and Documentation: A Retrospective","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Traceability; Documentation; Computer science; Code (set theory); Programming language; Software engineering","score_opus":0.011972415143162336,"score_gpt":0.26600079791641557,"score_spread":0.25402838277325324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406858280","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4956492,0.2109145,0.19591916,0.018602502,0.0027169255,0.0010610844,0.009593144,0.0029889059,0.06255454],"genre_scores_gemma":[0.6921091,0.11041187,0.14885843,0.0030844104,0.0012442343,0.0005223668,0.01417673,0.0021460808,0.027446726],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9832908,0.0049704425,0.002133326,0.0027862585,0.0063447184,0.00047437247],"domain_scores_gemma":[0.77048224,0.09430622,0.01879785,0.032174177,0.08202225,0.0022172327],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.028283417,0.0007125455,0.00068506587,0.017758086,0.0020590709,0.0064904615,0.0017297675,0.0014219654,0.003059176],"category_scores_gemma":[0.1475353,0.0011739128,0.0006358514,0.011864676,0.005031139,0.009780777,0.0035896997,0.0039115823,0.0026608696],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000417273,0.0005323506,0.0407283,0.0017484303,0.00013760202,0.00041007533,0.0091915,0.001733067,0.0046542464,0.020186165,0.016810497,0.9034505],"study_design_scores_gemma":[0.000066398155,0.0012925798,0.09940371,0.0065810727,0.0004430494,0.0028037648,0.007398463,0.005175687,0.04463681,0.016609164,0.8152593,0.00033008048],"about_ca_topic_score_codex":0.0109115,"about_ca_topic_score_gemma":0.009553482,"teacher_disagreement_score":0.028283417,"about_ca_system_score_codex":0.0037814246,"about_ca_system_score_gemma":0.006361611,"threshold_uncertainty_score":0.14957881},"labels":[],"label_agreement":null},{"id":"W4407098050","doi":"10.1109/tse.2025.3535938","title":"One Sentence Can Kill the Bug: Auto-Replay Mobile App Crashes From One-Sentence Overviews","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Youth Innovation Promotion Association of the Chinese Academy of Sciences; National Natural Science Foundation of China","keywords":"Computer science; Sentence; Natural language processing; Artificial intelligence; Speech recognition","score_opus":0.012259137582422999,"score_gpt":0.23431890563110322,"score_spread":0.22205976804868022,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407098050","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42970482,0.0028546539,0.23570116,0.00079587667,0.00044620622,0.00060324225,0.010350471,0.31627762,0.003265874],"genre_scores_gemma":[0.7819307,0.00072451815,0.18546827,0.00044027006,0.00011750801,0.0003592619,0.021750709,0.0054971944,0.0037115356],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9985322,0.00040702178,0.00011310642,0.00040772633,0.0004609123,0.000079060235],"domain_scores_gemma":[0.99286205,0.004226751,0.00067321904,0.0010910187,0.00093538105,0.00021159388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014919256,0.0024795332,0.00072152494,0.0016780476,0.00034754007,0.0007065416,0.0018253895,0.0010775282,0.0021409397],"category_scores_gemma":[0.01430951,0.00068730145,0.0008956536,0.00046882386,0.00052625133,0.0015331556,0.0017766461,0.0011736081,0.0013820608],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038010306,0.0007754678,0.053614985,0.0034433734,0.0006780685,0.008921466,0.006443066,0.074568726,0.08686287,0.002464058,0.08698018,0.67144674],"study_design_scores_gemma":[0.0003589569,0.0014504741,0.02858065,0.00025488372,0.00047426482,0.0030669838,0.0013236802,0.83822256,0.08421553,0.005286996,0.036460083,0.0003049657],"about_ca_topic_score_codex":0.0058122613,"about_ca_topic_score_gemma":0.00856003,"teacher_disagreement_score":0.0058122613,"about_ca_system_score_codex":0.00034384994,"about_ca_system_score_gemma":0.0008508968,"threshold_uncertainty_score":0.011556864},"labels":[],"label_agreement":null},{"id":"W4407361541","doi":"10.1109/tse.2025.3540549","title":"Search-Based DNN Testing and Retraining With GAN-Enhanced Simulations","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Parallel Computing and Optimization Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; European Space Agency; Science Foundation Ireland; Université du Luxembourg","keywords":"Computer science; Retraining; Artificial intelligence; Machine learning; Distributed computing; Computer architecture; Computer engineering","score_opus":0.01589608177396644,"score_gpt":0.24175184292689342,"score_spread":0.225855761152927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407361541","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2680368,0.00096084405,0.71447915,0.00070493214,0.00021043391,0.00020280159,0.0003331595,0.0064506237,0.008621193],"genre_scores_gemma":[0.9200156,0.00012160823,0.07733886,0.00023312191,0.000017845503,0.00012742537,0.00032272656,0.00024767464,0.0015751448],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993814,0.00021005594,0.00003624026,0.00013797366,0.00014821046,0.00008612198],"domain_scores_gemma":[0.99838793,0.0009600802,0.0001106306,0.00022704712,0.00025268111,0.00006162032],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013311933,0.0013823392,0.00068276864,0.00044333845,0.0002489931,0.0007100029,0.0019152099,0.0010638396,0.0020064437],"category_scores_gemma":[0.0050273333,0.00043704564,0.0006487095,0.00026619577,0.00078665104,0.0010395016,0.000899761,0.0014801226,0.00042055675],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008494724,0.000038018377,0.00086610584,0.00004404984,0.000025148896,0.000080696955,0.00003147503,0.9720638,0.0031176733,0.0011073769,0.000597851,0.021942863],"study_design_scores_gemma":[0.0000043869136,0.000019125724,0.00006957575,0.0000044369917,0.000003021685,0.0000109494085,0.000003626724,0.9971687,0.0020126335,0.000573066,0.00012793303,0.0000025615402],"about_ca_topic_score_codex":0.007453648,"about_ca_topic_score_gemma":0.0076787956,"teacher_disagreement_score":0.007453648,"about_ca_system_score_codex":0.0011833891,"about_ca_system_score_gemma":0.0009585945,"threshold_uncertainty_score":0.014820516},"labels":[],"label_agreement":null},{"id":"W4407375771","doi":"10.1109/tse.2025.3541166","title":"Automated Test Case Repair Using Language Models","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of Ottawa","funders":"","keywords":"Computer science; Programming language; Test (biology); Software engineering; Model-based testing; Reliability engineering; Test case; Natural language processing; Machine learning","score_opus":0.016937912449282302,"score_gpt":0.264907490693648,"score_spread":0.2479695782443657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407375771","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035338785,0.00016134627,0.9407604,0.00024963004,0.000050234055,0.00014794392,0.000249248,0.02088972,0.0021527058],"genre_scores_gemma":[0.6549723,0.00012349764,0.340464,0.0001225707,0.000020667205,0.00020411245,0.0006969121,0.0015590186,0.001836851],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974821,0.0009828376,0.000154993,0.00034893522,0.00079894,0.00023221651],"domain_scores_gemma":[0.9899734,0.0062434985,0.0007479583,0.0018214639,0.0010640683,0.00014961469],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014084746,0.0013323956,0.0009020319,0.0015323223,0.00050459104,0.0017537269,0.0021241787,0.0011914746,0.004174627],"category_scores_gemma":[0.010534418,0.00081021787,0.0015802844,0.0006877905,0.00075698126,0.0026998613,0.0015365522,0.0012293145,0.0011456972],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008766106,0.0006637577,0.005510743,0.0005797607,0.00023226097,0.0012904131,0.000620761,0.488429,0.038106255,0.037867084,0.01009853,0.41572475],"study_design_scores_gemma":[0.00004460781,0.000058524514,0.00017095827,0.00002826341,0.000042109707,0.000099851924,0.000044860833,0.97653174,0.008725635,0.012571926,0.0016633746,0.000018109242],"about_ca_topic_score_codex":0.005527335,"about_ca_topic_score_gemma":0.008360042,"teacher_disagreement_score":0.005527335,"about_ca_system_score_codex":0.0010517967,"about_ca_system_score_gemma":0.0021277415,"threshold_uncertainty_score":0.0139654875},"labels":[],"label_agreement":null},{"id":"W4408703448","doi":"10.1109/tse.2025.3553383","title":"Do Experts Agree About Smelly Infrastructure?","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Tracheal and airway disorders","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; McGill University","funders":"","keywords":"Computer science; Software engineering; Data science; Engineering management; Engineering","score_opus":0.006346365505415781,"score_gpt":0.23733527887889327,"score_spread":0.2309889133734775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408703448","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9228364,0.0012146359,0.012175115,0.020507893,0.00016825815,0.00009776334,0.00080997084,0.0002472737,0.041942626],"genre_scores_gemma":[0.9918637,0.00043476207,0.0020430426,0.0025609287,0.00007166521,0.000047347847,0.0005694669,0.00012051877,0.0022885357],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9790619,0.00760564,0.001261919,0.002841592,0.0076596397,0.0015692783],"domain_scores_gemma":[0.8768232,0.06122505,0.025988825,0.007847403,0.022577928,0.005537574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017705556,0.0003618406,0.00036048298,0.002970962,0.0021325366,0.004933856,0.0009567934,0.0018032275,0.00495959],"category_scores_gemma":[0.111211516,0.00045118178,0.00040435483,0.0023214754,0.0028520352,0.007238237,0.0035571253,0.0021272171,0.0014348044],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032226683,0.00013947564,0.5613431,0.00085872813,0.00016717182,0.0014911548,0.26084536,0.0010830642,0.007415489,0.008896237,0.028887533,0.12855045],"study_design_scores_gemma":[0.000053624914,0.00023600184,0.39774904,0.0011440031,0.00012437503,0.0015084863,0.43509033,0.005397818,0.0042672004,0.017546955,0.13653609,0.000346007],"about_ca_topic_score_codex":0.0070858663,"about_ca_topic_score_gemma":0.007835775,"teacher_disagreement_score":0.017705556,"about_ca_system_score_codex":0.0021145632,"about_ca_system_score_gemma":0.0018623911,"threshold_uncertainty_score":0.09363705},"labels":[],"label_agreement":null},{"id":"W4409771286","doi":"10.1109/tse.2025.3563121","title":"Testing CPS With Design Assumptions-Based Metamorphic Relations and Genetic Programming","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Evolutionary Algorithms and Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"HORIZON EUROPE European Innovation Council; Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg; Science Foundation Ireland","keywords":"Computer science; Genetic programming; Programming language; Software engineering; Artificial intelligence","score_opus":0.01636564723815365,"score_gpt":0.2155209431462641,"score_spread":0.19915529590811046,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409771286","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07920576,0.000094476985,0.91626513,0.00063467247,0.00002042133,0.00015806263,0.00007558714,0.0010253225,0.0025205412],"genre_scores_gemma":[0.55949396,0.00019259556,0.43816915,0.00033609322,0.000033390428,0.0002994505,0.00025368025,0.00016133788,0.0010603338],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9926116,0.003266515,0.0004288764,0.0012883175,0.0019537013,0.00045100908],"domain_scores_gemma":[0.97777474,0.01752826,0.002006635,0.0015468065,0.0009376369,0.00020597248],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044389125,0.0013716553,0.0006444028,0.0014228075,0.0004619307,0.001766617,0.001922481,0.0015423982,0.0011039538],"category_scores_gemma":[0.0274029,0.0006858266,0.0016689076,0.0008174268,0.0048361407,0.002521993,0.002185091,0.002384483,0.00015582595],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002160026,0.00025254276,0.006050607,0.00019855796,0.00009095623,0.0008308699,0.0007552606,0.7633412,0.008002294,0.15004088,0.0006573038,0.06956354],"study_design_scores_gemma":[0.00004308347,0.00013293482,0.0003477371,0.000051165815,0.000037637896,0.00013017144,0.00007274424,0.9252043,0.0066712857,0.06632603,0.0009566597,0.000026259953],"about_ca_topic_score_codex":0.0042478936,"about_ca_topic_score_gemma":0.0026984108,"teacher_disagreement_score":0.0044389125,"about_ca_system_score_codex":0.0016565244,"about_ca_system_score_gemma":0.0017790173,"threshold_uncertainty_score":0.023475468},"labels":[],"label_agreement":null},{"id":"W4409916786","doi":"10.1109/tse.2025.3565387","title":"Question Selection for Multimodal Code Search Synthesis Using Probabilistic Version Spaces","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Selection (genetic algorithm); Probabilistic logic; Modal; Programming language; Code (set theory); Theoretical computer science; Artificial intelligence; Set (abstract data type)","score_opus":0.01307841608488208,"score_gpt":0.2768731439809722,"score_spread":0.2637947278960901,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409916786","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010780149,0.00013201161,0.97758853,0.00018867693,0.00003236389,0.00017144534,0.0002073731,0.008015975,0.0028835987],"genre_scores_gemma":[0.30966115,0.00012976937,0.6815495,0.00019194059,0.00004816418,0.00059341313,0.0008937071,0.0014922147,0.005440101],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9939803,0.0026439042,0.00043420153,0.0013280434,0.0013415072,0.000272014],"domain_scores_gemma":[0.99019367,0.0074127596,0.00039685096,0.0010203886,0.0007538295,0.00022245852],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00431377,0.001236836,0.0012918463,0.002160809,0.0009557842,0.0027350802,0.0022634175,0.0019001684,0.019643089],"category_scores_gemma":[0.023985554,0.00080375647,0.0018171681,0.00088781555,0.0015101079,0.0043212483,0.004903532,0.0012323944,0.0031172985],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013125453,0.0002567356,0.0023565057,0.0008732995,0.00012934112,0.00072827825,0.0027930194,0.12976028,0.030256303,0.17278086,0.0090611065,0.64969176],"study_design_scores_gemma":[0.0000873324,0.00014137648,0.00027886813,0.00006746634,0.000042581214,0.00020511368,0.00027433297,0.88889617,0.016344894,0.08181558,0.011789792,0.000056557677],"about_ca_topic_score_codex":0.0016606309,"about_ca_topic_score_gemma":0.0018439629,"teacher_disagreement_score":0.019643089,"about_ca_system_score_codex":0.0012506358,"about_ca_system_score_gemma":0.0011594495,"threshold_uncertainty_score":0.06571269},"labels":[],"label_agreement":null},{"id":"W4410394325","doi":"10.1109/tse.2025.3570897","title":"Using Cooperative Co-Evolutionary Search to Generate Metamorphic Test Cases for Autonomous Driving Systems","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test (biology); Evolutionary algorithm; Artificial intelligence; Distributed computing; Geology","score_opus":0.047251793029374996,"score_gpt":0.3055320633810115,"score_spread":0.25828027035163653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410394325","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5143457,0.0007020927,0.47364137,0.00058379705,0.000092121845,0.00040999995,0.00027206857,0.0030503073,0.006902487],"genre_scores_gemma":[0.8718829,0.00009073766,0.125878,0.00016575055,0.000012936561,0.00020470525,0.0004193887,0.00015064553,0.0011949852],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99882287,0.00045203825,0.000055915923,0.00019125915,0.00033841268,0.00013950033],"domain_scores_gemma":[0.9953235,0.0033510365,0.00027598793,0.00034360585,0.0005356082,0.00017030971],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015311387,0.0012397239,0.00049570814,0.0015082519,0.00036191527,0.0006668704,0.0017683653,0.0011170364,0.0014803839],"category_scores_gemma":[0.0076241754,0.00041328144,0.0006888211,0.0005047108,0.0009592847,0.0007393013,0.0013001015,0.00085570256,0.00021181344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011911885,0.00024575574,0.00815287,0.00011368145,0.000064237065,0.0003885044,0.00013857262,0.8957475,0.007376323,0.0033722366,0.0011252244,0.083156005],"study_design_scores_gemma":[0.000018048104,0.00006983915,0.000291784,0.0000072045814,0.000009374249,0.000030757332,0.000022072756,0.9961909,0.0015291853,0.0014433576,0.0003832427,0.000004238321],"about_ca_topic_score_codex":0.0058220346,"about_ca_topic_score_gemma":0.0075561823,"teacher_disagreement_score":0.0058220346,"about_ca_system_score_codex":0.00082716433,"about_ca_system_score_gemma":0.0013233268,"threshold_uncertainty_score":0.011576295},"labels":[],"label_agreement":null},{"id":"W4411232430","doi":"10.1109/tse.2025.3579574","title":"BLAZE: Cross-Language and Cross-Project Bug Localization via Dynamic Chunking and Hard Example Learning","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; University of Waterloo","funders":"","keywords":"Computer science; Chunking (psychology); Programming language; Software engineering; Artificial intelligence; Natural language processing","score_opus":0.010397706086851791,"score_gpt":0.2798341540061944,"score_spread":0.26943644791934257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411232430","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03354417,0.00037777776,0.82306075,0.00073211757,0.00015343718,0.0003617528,0.0008240105,0.13839372,0.0025522944],"genre_scores_gemma":[0.15863477,0.00014828444,0.82728803,0.00056757027,0.00003934141,0.00038218836,0.0030003944,0.0051984577,0.004740914],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99599755,0.0010708267,0.00027367828,0.0013777579,0.0010051976,0.00027503516],"domain_scores_gemma":[0.9898369,0.0035196808,0.00092071656,0.0040808944,0.0011763826,0.00046540567],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046475027,0.00252505,0.0013192251,0.0022569345,0.00084734894,0.0018533549,0.0056094993,0.0020890029,0.004482385],"category_scores_gemma":[0.020092715,0.0016703074,0.0016423232,0.0013409283,0.0015376498,0.007209259,0.008826314,0.0038028175,0.0025363814],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075177324,0.0007254828,0.013602375,0.00044307145,0.00031309595,0.00050828146,0.0015434494,0.08985984,0.020729594,0.008526217,0.040288653,0.82270813],"study_design_scores_gemma":[0.00015235863,0.00031106337,0.0020734502,0.00004698885,0.000067112545,0.0002484189,0.0002824485,0.9424685,0.02002176,0.019357998,0.014877562,0.000092362345],"about_ca_topic_score_codex":0.007857844,"about_ca_topic_score_gemma":0.017470857,"teacher_disagreement_score":0.007857844,"about_ca_system_score_codex":0.0012381421,"about_ca_system_score_gemma":0.0029410685,"threshold_uncertainty_score":0.02457869},"labels":[],"label_agreement":null},{"id":"W4413822387","doi":"10.1109/tse.2025.3603897","title":"A Systematic Literature Review of Machine Learning Approaches for Migrating Monolithic Systems to Microservices","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Microservices; Computer science; Software engineering; Artificial intelligence; Data science; Machine learning; Programming language; Operating system; Cloud computing","score_opus":0.010631439464268127,"score_gpt":0.22710151134741663,"score_spread":0.2164700718831485,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413822387","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00066598953,0.99572206,0.0009441296,0.0008435899,0.00017294943,0.00038893294,0.00059773005,0.000025167386,0.0006393423],"genre_scores_gemma":[0.009364747,0.98358077,0.003906556,0.0010151052,0.000111562535,0.0012907967,0.0005468072,0.000020100944,0.00016351142],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.97526544,0.009375082,0.008802851,0.001982531,0.0040852386,0.0004888784],"domain_scores_gemma":[0.8629106,0.109617606,0.012588137,0.0027502833,0.011429532,0.00070378545],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025339967,0.0024813358,0.0063324585,0.029422117,0.0014312344,0.0048339795,0.0037677058,0.0028068905,0.006605148],"category_scores_gemma":[0.12648723,0.0016669926,0.010064528,0.020339744,0.0016655062,0.006349262,0.0032358156,0.0027635347,0.0010772403],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000101692545,0.000032557957,0.0008642486,0.8855702,0.0032598318,0.0001051504,0.000588772,0.00029212763,0.00016904439,0.0011806581,0.0034581765,0.10437758],"study_design_scores_gemma":[0.000048173468,0.00008968714,0.0017343531,0.9545656,0.012526275,0.0001851725,0.0004913843,0.00012444229,0.00019229311,0.00090341707,0.02909921,0.00003988719],"about_ca_topic_score_codex":0.014512237,"about_ca_topic_score_gemma":0.041700646,"teacher_disagreement_score":0.029422117,"about_ca_system_score_codex":0.008633699,"about_ca_system_score_gemma":0.034755338,"threshold_uncertainty_score":0.1340121},"labels":[],"label_agreement":null},{"id":"W4413925951","doi":"10.1109/tse.2025.3605442","title":"Towards Explainable Vulnerability Detection With Large Language Models","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Vulnerability (computing); Data science; Programming language; Software engineering; Natural language processing; Computer security","score_opus":0.00918862300238798,"score_gpt":0.22338367195039308,"score_spread":0.2141950489480051,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413925951","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029745858,0.00080626766,0.9376738,0.0013618723,0.000075198186,0.00017119067,0.0018959249,0.027323835,0.00094608066],"genre_scores_gemma":[0.2944218,0.000492443,0.69129163,0.00090100727,0.00012855537,0.00040123225,0.008626661,0.0014070394,0.0023296925],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99659675,0.0016586152,0.00017305942,0.0009257084,0.00048267806,0.00016307273],"domain_scores_gemma":[0.9851916,0.011303282,0.0008434893,0.001465416,0.00097595464,0.00022029906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034324524,0.0023672124,0.0009023211,0.0034088227,0.00064311834,0.001913844,0.0023934187,0.0022518332,0.0028467688],"category_scores_gemma":[0.022010129,0.0010111107,0.0024115671,0.0015717818,0.00113794,0.004958555,0.0037941278,0.0043412484,0.0021028896],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000505737,0.0003813963,0.016500073,0.0012467272,0.00042583267,0.0010940185,0.0024576995,0.18118477,0.02319397,0.01807875,0.03401882,0.7209122],"study_design_scores_gemma":[0.000037087117,0.0000553896,0.00096457935,0.00006226759,0.00006662579,0.00016100511,0.00021220978,0.9524358,0.005537703,0.033696212,0.006734546,0.000036645775],"about_ca_topic_score_codex":0.0050548757,"about_ca_topic_score_gemma":0.011988171,"teacher_disagreement_score":0.0050548757,"about_ca_system_score_codex":0.0013328749,"about_ca_system_score_gemma":0.0020748284,"threshold_uncertainty_score":0.018152773},"labels":[],"label_agreement":null},{"id":"W4414165847","doi":"10.1109/tse.2025.3603009","title":"Identifying Reusable Services in Legacy Object-Oriented Systems: A Type-Sensitive Identification Approach","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Service-Oriented Architecture and Web Services","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; École de Technologie Supérieure","funders":"Canada Research Chairs","keywords":"Identification (biology); Legacy system; Service (business); sync; Service-oriented architecture; Software; Software maintenance","score_opus":0.00791555203849998,"score_gpt":0.22490668429062846,"score_spread":0.2169911322521285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414165847","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010192983,0.00037294513,0.9834915,0.0010868455,0.00003734119,0.0002345125,0.00006850936,0.0006928511,0.00382256],"genre_scores_gemma":[0.13366689,0.0006943637,0.8597491,0.0005654302,0.0000679773,0.0002703949,0.00028358528,0.00022874046,0.00447348],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9875267,0.0039527025,0.0011269709,0.0015352111,0.005110406,0.00074798986],"domain_scores_gemma":[0.97854155,0.009939862,0.0021150024,0.0038769683,0.0050221602,0.00050449953],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014179169,0.0010697631,0.0012076263,0.0083512915,0.0027240925,0.007164108,0.0041590687,0.0032987124,0.001340849],"category_scores_gemma":[0.02388592,0.001414748,0.0031434274,0.003960861,0.0038727615,0.010991369,0.004557344,0.0032548464,0.0009153778],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018947711,0.00041763374,0.015136093,0.00088586763,0.00017717977,0.0015253697,0.008860848,0.04058334,0.013968176,0.42721948,0.005243459,0.48579314],"study_design_scores_gemma":[0.000057000118,0.00016945075,0.004895375,0.0009496922,0.00041494163,0.0023655326,0.0059194737,0.47961903,0.03178332,0.40433523,0.06917946,0.0003114151],"about_ca_topic_score_codex":0.0072165355,"about_ca_topic_score_gemma":0.0059281653,"teacher_disagreement_score":0.014179169,"about_ca_system_score_codex":0.004086129,"about_ca_system_score_gemma":0.005737562,"threshold_uncertainty_score":0.07498753},"labels":[],"label_agreement":null},{"id":"W4414322174","doi":"10.1109/tse.2025.3610540","title":"Diagnosing Unknown Attacks in Smart Homes Using Abductive Reasoning","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Network Security and Intrusion Detection","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trinity College","funders":"Science Foundation Ireland","keywords":"Smart card; Key (lock); Abductive reasoning; Data integrity; Home automation","score_opus":0.009391897563815026,"score_gpt":0.23669312742569332,"score_spread":0.22730122986187828,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414322174","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15293756,0.00042377246,0.8402952,0.00066623226,0.00006503895,0.00015158842,0.00028167962,0.0023711172,0.0028078558],"genre_scores_gemma":[0.8755987,0.00015028584,0.12292714,0.00009234196,0.000021555792,0.000042545424,0.00022936822,0.000037709193,0.0009002504],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99849236,0.00038506667,0.00015164065,0.0003590784,0.00041963125,0.0001922758],"domain_scores_gemma":[0.99659157,0.0023450048,0.00034325197,0.0003040677,0.00033616842,0.00007999364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014833291,0.0012401703,0.0010063987,0.0014036421,0.0006513055,0.0015690723,0.0017269974,0.0017072578,0.0016788288],"category_scores_gemma":[0.005864488,0.00069865334,0.0011160626,0.0005836127,0.001232084,0.0022392983,0.0015002927,0.0014918736,0.0002602264],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048405505,0.00031559012,0.012310112,0.00024823783,0.0002168409,0.001230261,0.00033804576,0.8618372,0.0062853997,0.009078521,0.0016201965,0.106035605],"study_design_scores_gemma":[0.000019030533,0.000023813724,0.0003780938,0.000013851835,0.000024640472,0.000062356485,0.000068472524,0.98977494,0.0015933523,0.007854152,0.00018026453,0.000007051297],"about_ca_topic_score_codex":0.0087465625,"about_ca_topic_score_gemma":0.01149691,"teacher_disagreement_score":0.0087465625,"about_ca_system_score_codex":0.00093720295,"about_ca_system_score_gemma":0.0010414534,"threshold_uncertainty_score":0.017391324},"labels":[],"label_agreement":null},{"id":"W4414348545","doi":"10.1109/tse.2025.3611329","title":"DiffGAN: A Test Generation Approach for Differential Testing of Deep Neural Networks for Image Analysis","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Image Processing and 3D Reconstruction","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Science Foundation Ireland; General Motors Corporation","keywords":"Artificial neural network; Image (mathematics); Pattern recognition (psychology); Test (biology); Image processing; Differential (mechanical device)","score_opus":0.01220046323444045,"score_gpt":0.22245420945146246,"score_spread":0.210253746217022,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414348545","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012034551,0.00013755448,0.97469145,0.0001331554,0.000058667913,0.000115994866,0.0003699989,0.01163716,0.0008215903],"genre_scores_gemma":[0.2945056,0.000082103266,0.69754815,0.00044168156,0.00008348875,0.0005985071,0.0021037224,0.0023490551,0.0022877117],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99753785,0.0011102307,0.00016265488,0.00042682592,0.00057483045,0.00018760211],"domain_scores_gemma":[0.9913775,0.0063035493,0.00030222084,0.00089921866,0.00095001963,0.00016751932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036135712,0.0017632374,0.00096826965,0.0018365765,0.0003821569,0.0008481131,0.003907755,0.0017869711,0.008573485],"category_scores_gemma":[0.01525741,0.00082630996,0.0013526903,0.0006572013,0.001112938,0.001775204,0.0022943649,0.002424213,0.0012929332],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014090834,0.00035829755,0.0051954454,0.00052620494,0.00038688074,0.000658726,0.00016632678,0.3658089,0.02125873,0.026537936,0.019896965,0.55779654],"study_design_scores_gemma":[0.00007483416,0.000104620376,0.0002137623,0.000017169543,0.000019729969,0.000080216225,0.000010782693,0.97803986,0.008401519,0.012006849,0.0010188754,0.000011742762],"about_ca_topic_score_codex":0.0025218127,"about_ca_topic_score_gemma":0.0038356688,"teacher_disagreement_score":0.008573485,"about_ca_system_score_codex":0.0011338445,"about_ca_system_score_gemma":0.0015071399,"threshold_uncertainty_score":0.0286811},"labels":[],"label_agreement":null},{"id":"W4414404854","doi":"10.1109/tse.2025.3612253","title":"MetaSel: A Test Selection Approach for Fine-Tuned DNN Models","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Ottawa","funders":"Alliance de recherche numérique du Canada; Mitacs; Huawei Technologies","keywords":"Covariate; Software deployment; Model selection; Selection (genetic algorithm); Context (archaeology); Subspace topology; Statistical hypothesis testing; Test data; Probability distribution","score_opus":0.01927186097653321,"score_gpt":0.24144192629660027,"score_spread":0.22217006532006706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414404854","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10118463,0.0024629848,0.8488248,0.0005733983,0.00024944465,0.0005115588,0.0016282456,0.040826585,0.0037383656],"genre_scores_gemma":[0.55665493,0.00042849724,0.42513007,0.001576354,0.00016937207,0.0007668921,0.007639664,0.003337156,0.0042970343],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969812,0.001019557,0.0002531388,0.0009037023,0.0006214065,0.00022089604],"domain_scores_gemma":[0.9905934,0.005397995,0.0006125016,0.0015138785,0.001565166,0.0003171171],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054090726,0.0036940724,0.001785994,0.0027669722,0.0006798228,0.0016823554,0.0051355194,0.002027119,0.0038030974],"category_scores_gemma":[0.020582594,0.0011162322,0.0016467323,0.0010531729,0.0009281341,0.003077069,0.0031397452,0.0031946886,0.0019736588],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009997992,0.00056040747,0.024298042,0.0005042787,0.00071890384,0.00067388854,0.00025380278,0.35096568,0.02146559,0.0034125322,0.016282888,0.5798642],"study_design_scores_gemma":[0.00008312088,0.00024267017,0.001281036,0.00004265992,0.00008688369,0.00014830157,0.00006121718,0.9818594,0.008856926,0.005247846,0.002056417,0.00003348927],"about_ca_topic_score_codex":0.0060395906,"about_ca_topic_score_gemma":0.014204347,"teacher_disagreement_score":0.0060395906,"about_ca_system_score_codex":0.0014696501,"about_ca_system_score_gemma":0.0023130598,"threshold_uncertainty_score":0.028606296},"labels":[],"label_agreement":null},{"id":"W4415481257","doi":"10.1109/tse.2025.3624631","title":"Contrasting the Hyperparameter Tuning Impact Across Software Defect Prediction Scenarios","year":2025,"lang":"","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Hyperparameter; Software; Hyperparameter optimization; Software quality assurance; Software bug; Scope (computer science); Quality (philosophy)","score_opus":0.015967407007470576,"score_gpt":0.2777652525143255,"score_spread":0.2617978455068549,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415481257","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94948083,0.002195869,0.042347077,0.00077234977,0.00015871933,0.00022742714,0.0006589225,0.0017874922,0.0023713757],"genre_scores_gemma":[0.978438,0.00029571494,0.019380337,0.0001830628,0.00004300483,0.0001160037,0.0011577134,0.00011334306,0.0002727956],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9899809,0.004839719,0.00097517134,0.0021251643,0.0014935879,0.00058548816],"domain_scores_gemma":[0.9223199,0.061142135,0.0037346093,0.007933238,0.003996712,0.00087334757],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013249129,0.002026124,0.00082761806,0.0018767365,0.0005737123,0.0017912093,0.0013341994,0.0019277601,0.00045641727],"category_scores_gemma":[0.06672763,0.0005084647,0.0009259802,0.0013718787,0.00103117,0.0029498672,0.0013793211,0.0024071592,0.00024327531],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021534476,0.0013327636,0.074624345,0.00045149843,0.0007635978,0.0002615709,0.00029085297,0.7597412,0.0134532275,0.0017953288,0.0039902492,0.14114186],"study_design_scores_gemma":[0.0002486484,0.0018507919,0.03589935,0.00010853346,0.00027501304,0.00024598508,0.0003737556,0.93270034,0.023073275,0.0034915837,0.0016254337,0.00010728609],"about_ca_topic_score_codex":0.0036198723,"about_ca_topic_score_gemma":0.0025767598,"teacher_disagreement_score":0.013249129,"about_ca_system_score_codex":0.0009889441,"about_ca_system_score_gemma":0.0009017509,"threshold_uncertainty_score":0.070068955},"labels":[],"label_agreement":null},{"id":"W4415821354","doi":"10.1109/tse.2025.3627891","title":"Causes and Canonicalization of Unreproducible Builds in Java","year":2025,"lang":"","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Artifact (error); Java; Software; Taxonomy (biology); Focus (optics); Identification (biology); Software development; Legacy system","score_opus":0.014029697954729706,"score_gpt":0.2637907972674899,"score_spread":0.2497610993127602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415821354","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9569225,0.0015486641,0.017020237,0.00062072615,0.00005610615,0.00014519556,0.015708236,0.0037611942,0.0042172475],"genre_scores_gemma":[0.9254166,0.0010799365,0.019269552,0.00024552274,0.00005050439,0.00024261558,0.0507984,0.0011352141,0.0017616496],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9867597,0.0027223215,0.0018995146,0.0027361163,0.0050926832,0.00078962144],"domain_scores_gemma":[0.8934863,0.047383662,0.022436693,0.026514603,0.008815219,0.0013634969],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004678768,0.00048817418,0.00044648562,0.0066070217,0.0013850775,0.0017079617,0.0011554394,0.0009705625,0.0009034631],"category_scores_gemma":[0.047979247,0.0005316551,0.0008102873,0.007904416,0.0014749378,0.0017982329,0.002419295,0.0012667907,0.0005138615],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003105933,0.00019537158,0.8580597,0.0007333464,0.00015980443,0.001815995,0.0046854517,0.006200124,0.004820467,0.0062597333,0.018227087,0.09853227],"study_design_scores_gemma":[0.000055578257,0.0001526745,0.85151976,0.00048473378,0.00018169961,0.0046070814,0.00405222,0.03133981,0.015487609,0.0075559043,0.08443196,0.00013093825],"about_ca_topic_score_codex":0.01060992,"about_ca_topic_score_gemma":0.01861179,"teacher_disagreement_score":0.9953212,"about_ca_system_score_codex":0.0010630742,"about_ca_system_score_gemma":0.0025864055,"threshold_uncertainty_score":0.024743974},"labels":[],"label_agreement":null},{"id":"W4417439110","doi":"10.1109/tse.2025.3645143","title":"CARE: Context Aware Root Cause Identification Using Distributed Traces and Profiling Metrics","year":2025,"lang":"","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Brock University; University of Alberta","funders":"","keywords":"Root cause; Observability; Root cause analysis; Profiling (computer programming); Root (linguistics); Identification (biology); Benchmark (surveying)","score_opus":0.01619043411345925,"score_gpt":0.2561327355253319,"score_spread":0.23994230141187264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417439110","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10325367,0.0017976305,0.8481811,0.0010005683,0.00027494197,0.00061170565,0.003388509,0.037845593,0.0036462122],"genre_scores_gemma":[0.75604594,0.00057225936,0.2353939,0.0003182645,0.00012176984,0.00029120833,0.0047422494,0.0006090641,0.001905265],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99691975,0.0006260702,0.00025373272,0.0008315558,0.0011536969,0.00021525146],"domain_scores_gemma":[0.99134994,0.0030754355,0.0015696266,0.0016914452,0.001856445,0.00045705098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002201068,0.00279651,0.001294826,0.005943173,0.000741349,0.0018472096,0.001997745,0.0012212942,0.001159078],"category_scores_gemma":[0.014179982,0.00044836133,0.00085476774,0.00224472,0.00060892175,0.0025251326,0.0026306165,0.0016274228,0.00085338793],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004995065,0.0008313755,0.16763635,0.0009403484,0.0006922047,0.0012978646,0.0009440338,0.1481529,0.023074694,0.0072746417,0.028760102,0.619896],"study_design_scores_gemma":[0.000050533567,0.00026351138,0.018089615,0.00012367786,0.0001403369,0.0007875201,0.0006700643,0.9430099,0.012758361,0.016082825,0.007934905,0.00008878243],"about_ca_topic_score_codex":0.008404993,"about_ca_topic_score_gemma":0.01559939,"teacher_disagreement_score":0.008404993,"about_ca_system_score_codex":0.00085531856,"about_ca_system_score_gemma":0.0023463727,"threshold_uncertainty_score":0.01671213},"labels":[],"label_agreement":null}]}